Prechádzať zdrojové kódy

fix: stop retrying refusals the provider says will never change

The ten OpenAI-compatible chat nodes retried every failure alike. A wrong key,
a spent allowance and a billing stop were each asked three times with backoff
in between - which delays the report and, when the refusal is an exhausted
quota, spends more of whatever ran out. ollama-chat was fixed for this in
August; its nine siblings were not.

The policy is per provider because their documentation genuinely disagrees, and
not in small ways:

- Groq documents 422 as worth retrying - it covers semantic errors and model
  hallucinations. Mistral documents the same 422 as a permanent validation
  error. Same status, opposite correct action.
- Fireworks documents 500 as permanent, "a server-side code bug unlikely to
  resolve on its own" - alone among all ten in saying so about the one status
  everyone assumes is transient.
- "Out of money" arrives four different ways: 402 (OpenRouter, Together,
  DeepSeek, Fireworks), 429 with an error code (OpenAI), 401 (Perplexity) and
  403 (Ollama, which is what started this). Keying on any single status misses
  most of them.
- Together's 403 is not a permissions error at all: it means the prompt
  exceeded the model's context length.
- OpenAI's 429 is split. "Rate limit reached" is transient, but an exhausted
  credit balance and a project spend limit arrive as 429 too, and no amount of
  backing off clears them - so the error code is checked as well as the status.

Each provider entry carries a note saying what its docs say, so the next person
changing one knows whether they are correcting a mistake or contradicting a
vendor. xAI publishes no error table; it gets the conservative set the others
agree on, and says so.

Two tests pin it, both hitting the same non-existent local endpoint and its
identical 404: groq-chat stops at one attempt, openai-chat uses all three,
because Groq documents 404 as permanent and OpenAI does not. The pair fails if
anyone collapses these back into one shared list.

verify-node.py grew credential support to make that testable. Cases declare
`credentials` and reference them as {{credential:key}}, created before the run
and deleted after - the one existing case that needed a credential names it by
id, which ties it to a single installation.
fszontagh 3 týždňov pred
rodič
commit
a1d077ca04

+ 27 - 2
nodes/ai/deepinfra-chat.js

@@ -4,7 +4,7 @@
  * @category ai
  * @version 1.0.0
  * @description Ask DeepInfra a question, with an image if there is one, and optionally get JSON back
- * @icon server
+ * @icon message-square
  */
 
 // Generated by scripts/gen-openai-compatible-nodes.py from one implementation
@@ -179,6 +179,15 @@ const outputSchema = {
 
 const PROVIDER = 'DeepInfra';
 
+// Statuses this provider documents as answers that will not change.
+//
+// DeepInfra documents little beyond 429, which is engine_overloaded and transient (a rejected request is not billed). The 4xx set is the conservative common one.
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = [400, 401, 402, 403, 404];
+const PERMANENT_ERROR_CODES = [];
+
 function baseOf(config) {
     const value = String(config.baseUrl || 'https://api.deepinfra.com/v1/openai').trim().replace(/\/+$/, '');
     if (!value) {
@@ -228,13 +237,24 @@ function request(options) {
 
     if (response.status < 200 || response.status >= 300) {
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         } else if (typeof body === 'string' && body) {
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }
 
     return body || {};
@@ -506,6 +526,11 @@ async function execute(config, input, context) {
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }
             if (attempt < attempts) {
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;

+ 27 - 2
nodes/ai/deepseek-chat.js

@@ -4,7 +4,7 @@
  * @category ai
  * @version 1.0.0
  * @description Ask DeepSeek a question, with an image if there is one, and optionally get JSON back
- * @icon deepseek
+ * @icon message-square
  */
 
 // Generated by scripts/gen-openai-compatible-nodes.py from one implementation
@@ -179,6 +179,15 @@ const outputSchema = {
 
 const PROVIDER = 'DeepSeek';
 
+// Statuses this provider documents as answers that will not change.
+//
+// 402 is Insufficient Balance. 422 is Invalid Parameters and permanent here - note Groq treats the same status as retryable.
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = [400, 401, 402, 422];
+const PERMANENT_ERROR_CODES = [];
+
 function baseOf(config) {
     const value = String(config.baseUrl || 'https://api.deepseek.com/v1').trim().replace(/\/+$/, '');
     if (!value) {
@@ -228,13 +237,24 @@ function request(options) {
 
     if (response.status < 200 || response.status >= 300) {
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         } else if (typeof body === 'string' && body) {
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }
 
     return body || {};
@@ -506,6 +526,11 @@ async function execute(config, input, context) {
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }
             if (attempt < attempts) {
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;

+ 27 - 2
nodes/ai/fireworks-chat.js

@@ -4,7 +4,7 @@
  * @category ai
  * @version 1.0.0
  * @description Ask Fireworks AI a question, with an image if there is one, and optionally get JSON back
- * @icon flame
+ * @icon message-square
  */
 
 // Generated by scripts/gen-openai-compatible-nodes.py from one implementation
@@ -179,6 +179,15 @@ const outputSchema = {
 
 const PROVIDER = 'Fireworks AI';
 
+// Statuses this provider documents as answers that will not change.
+//
+// 500 is permanent here, alone among these providers: Fireworks documents it as a server-side code bug unlikely to resolve on its own. 402 is a billing stop. 502 and 503 stay retryable.
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = [400, 401, 402, 403, 404, 500];
+const PERMANENT_ERROR_CODES = [];
+
 function baseOf(config) {
     const value = String(config.baseUrl || 'https://api.fireworks.ai/inference/v1').trim().replace(/\/+$/, '');
     if (!value) {
@@ -228,13 +237,24 @@ function request(options) {
 
     if (response.status < 200 || response.status >= 300) {
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         } else if (typeof body === 'string' && body) {
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }
 
     return body || {};
@@ -506,6 +526,11 @@ async function execute(config, input, context) {
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }
             if (attempt < attempts) {
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;

+ 27 - 2
nodes/ai/groq-chat.js

@@ -4,7 +4,7 @@
  * @category ai
  * @version 1.0.0
  * @description Ask Groq a question, with an image if there is one, and optionally get JSON back
- * @icon zap
+ * @icon message-square
  */
 
 // Generated by scripts/gen-openai-compatible-nodes.py from one implementation
@@ -179,6 +179,15 @@ const outputSchema = {
 
 const PROVIDER = 'Groq';
 
+// Statuses this provider documents as answers that will not change.
+//
+// 413 (payload too large) and 499 (caller cancelled) are permanent. 422 is deliberately absent: Groq documents it as worth retrying, because it covers semantic errors and model hallucinations. 498 is flex-tier capacity and explicitly says to try again later.
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = [400, 401, 403, 404, 413, 499];
+const PERMANENT_ERROR_CODES = [];
+
 function baseOf(config) {
     const value = String(config.baseUrl || 'https://api.groq.com/openai/v1').trim().replace(/\/+$/, '');
     if (!value) {
@@ -228,13 +237,24 @@ function request(options) {
 
     if (response.status < 200 || response.status >= 300) {
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         } else if (typeof body === 'string' && body) {
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }
 
     return body || {};
@@ -506,6 +526,11 @@ async function execute(config, input, context) {
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }
             if (attempt < attempts) {
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;

+ 27 - 2
nodes/ai/mistral-chat.js

@@ -4,7 +4,7 @@
  * @category ai
  * @version 1.0.0
  * @description Ask Mistral a question, with an image if there is one, and optionally get JSON back
- * @icon mistral
+ * @icon message-square
  */
 
 // Generated by scripts/gen-openai-compatible-nodes.py from one implementation
@@ -179,6 +179,15 @@ const outputSchema = {
 
 const PROVIDER = 'Mistral';
 
+// Statuses this provider documents as answers that will not change.
+//
+// 422 is a validation error and permanent - the opposite of Groq's reading of the same code.
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = [400, 401, 403, 404, 422];
+const PERMANENT_ERROR_CODES = [];
+
 function baseOf(config) {
     const value = String(config.baseUrl || 'https://api.mistral.ai/v1').trim().replace(/\/+$/, '');
     if (!value) {
@@ -228,13 +237,24 @@ function request(options) {
 
     if (response.status < 200 || response.status >= 300) {
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         } else if (typeof body === 'string' && body) {
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }
 
     return body || {};
@@ -506,6 +526,11 @@ async function execute(config, input, context) {
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }
             if (attempt < attempts) {
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;

+ 27 - 2
nodes/ai/openai-chat.js

@@ -4,7 +4,7 @@
  * @category ai
  * @version 1.0.0
  * @description Ask OpenAI a question, with an image if there is one, and optionally get JSON back
- * @icon sparkles
+ * @icon message-square
  */
 
 // Generated by scripts/gen-openai-compatible-nodes.py from one implementation
@@ -179,6 +179,15 @@ const outputSchema = {
 
 const PROVIDER = 'OpenAI';
 
+// Statuses this provider documents as answers that will not change.
+//
+// OpenAI splits 429: "rate limit reached" is transient, but an exhausted credit balance, a project spend limit and an organisation usage limit all arrive as 429 too and no amount of backing off clears them. The error code tells them apart, which is why permanentErrorCodes exists.
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = [400, 401, 403];
+const PERMANENT_ERROR_CODES = ['insufficient_quota', 'billing_hard_limit_reached', 'access_terminated'];
+
 function baseOf(config) {
     const value = String(config.baseUrl || 'https://api.openai.com/v1').trim().replace(/\/+$/, '');
     if (!value) {
@@ -228,13 +237,24 @@ function request(options) {
 
     if (response.status < 200 || response.status >= 300) {
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         } else if (typeof body === 'string' && body) {
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }
 
     return body || {};
@@ -506,6 +526,11 @@ async function execute(config, input, context) {
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }
             if (attempt < attempts) {
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;

+ 27 - 2
nodes/ai/openrouter-chat.js

@@ -4,7 +4,7 @@
  * @category ai
  * @version 1.0.0
  * @description Ask OpenRouter a question, with an image if there is one, and optionally get JSON back
- * @icon openrouter
+ * @icon message-square
  */
 
 // Generated by scripts/gen-openai-compatible-nodes.py from one implementation
@@ -179,6 +179,15 @@ const outputSchema = {
 
 const PROVIDER = 'OpenRouter';
 
+// Statuses this provider documents as answers that will not change.
+//
+// 402 is OpenRouter's own: insufficient credits. 403 covers moderation and guardrail blocks, which the same request will always trip. 408/502/503 are transient - it may reroute to another provider.
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = [400, 401, 402, 403];
+const PERMANENT_ERROR_CODES = [];
+
 function baseOf(config) {
     const value = String(config.baseUrl || 'https://openrouter.ai/api/v1').trim().replace(/\/+$/, '');
     if (!value) {
@@ -228,13 +237,24 @@ function request(options) {
 
     if (response.status < 200 || response.status >= 300) {
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         } else if (typeof body === 'string' && body) {
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }
 
     return body || {};
@@ -506,6 +526,11 @@ async function execute(config, input, context) {
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }
             if (attempt < attempts) {
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;

+ 27 - 2
nodes/ai/perplexity-chat.js

@@ -4,7 +4,7 @@
  * @category ai
  * @version 1.0.0
  * @description Ask Perplexity a question, with an image if there is one, and optionally get JSON back
- * @icon perplexity
+ * @icon message-square
  */
 
 // Generated by scripts/gen-openai-compatible-nodes.py from one implementation
@@ -171,6 +171,15 @@ const outputSchema = {
 
 const PROVIDER = 'Perplexity';
 
+// Statuses this provider documents as answers that will not change.
+//
+// 401 does double duty: an invalid key AND an account out of credits both return it, so a 401 here is not necessarily a wrong key.
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = [400, 401, 403];
+const PERMANENT_ERROR_CODES = [];
+
 function baseOf(config) {
     const value = String(config.baseUrl || 'https://api.perplexity.ai').trim().replace(/\/+$/, '');
     if (!value) {
@@ -220,13 +229,24 @@ function request(options) {
 
     if (response.status < 200 || response.status >= 300) {
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         } else if (typeof body === 'string' && body) {
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }
 
     return body || {};
@@ -480,6 +500,11 @@ async function execute(config, input, context) {
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }
             if (attempt < attempts) {
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;

+ 27 - 2
nodes/ai/together-chat.js

@@ -4,7 +4,7 @@
  * @category ai
  * @version 1.0.0
  * @description Ask Together AI a question, with an image if there is one, and optionally get JSON back
- * @icon users
+ * @icon message-square
  */
 
 // Generated by scripts/gen-openai-compatible-nodes.py from one implementation
@@ -179,6 +179,15 @@ const outputSchema = {
 
 const PROVIDER = 'Together AI';
 
+// Statuses this provider documents as answers that will not change.
+//
+// 402 is the monthly spending limit. 403 is NOT a permissions error here - it means the prompt exceeded the model's context length, which retrying cannot shorten.
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = [400, 401, 402, 403, 404];
+const PERMANENT_ERROR_CODES = [];
+
 function baseOf(config) {
     const value = String(config.baseUrl || 'https://api.together.xyz/v1').trim().replace(/\/+$/, '');
     if (!value) {
@@ -228,13 +237,24 @@ function request(options) {
 
     if (response.status < 200 || response.status >= 300) {
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         } else if (typeof body === 'string' && body) {
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }
 
     return body || {};
@@ -506,6 +526,11 @@ async function execute(config, input, context) {
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }
             if (attempt < attempts) {
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;

+ 27 - 2
nodes/ai/xai-chat.js

@@ -4,7 +4,7 @@
  * @category ai
  * @version 1.0.0
  * @description Ask xAI a question, with an image if there is one, and optionally get JSON back
- * @icon xai
+ * @icon message-square
  */
 
 // Generated by scripts/gen-openai-compatible-nodes.py from one implementation
@@ -179,6 +179,15 @@ const outputSchema = {
 
 const PROVIDER = 'xAI';
 
+// Statuses this provider documents as answers that will not change.
+//
+// xAI publishes no error-code table. This is the conservative common set every OpenAI-compatible provider agrees on; nothing provider-specific is assumed.
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = [400, 401, 403, 404];
+const PERMANENT_ERROR_CODES = [];
+
 function baseOf(config) {
     const value = String(config.baseUrl || 'https://api.x.ai/v1').trim().replace(/\/+$/, '');
     if (!value) {
@@ -228,13 +237,24 @@ function request(options) {
 
     if (response.status < 200 || response.status >= 300) {
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         } else if (typeof body === 'string' && body) {
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }
 
     return body || {};
@@ -506,6 +526,11 @@ async function execute(config, input, context) {
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }
             if (attempt < attempts) {
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;

+ 69 - 11
scripts/gen-openai-compatible-nodes.py

@@ -19,63 +19,90 @@ from pathlib import Path
 
 PROVIDERS = [
     {
-        'id': 'openai', 'name': 'OpenAI',
+        'id': 'openai',
+        'permanent_statuses': [400, 401, 403],
+        'permanent_error_codes': ['insufficient_quota', 'billing_hard_limit_reached', 'access_terminated'],
+        'retry_note': 'OpenAI splits 429: "rate limit reached" is transient, but an exhausted credit balance, a project spend limit and an organisation usage limit all arrive as 429 too and no amount of backing off clears them. The error code tells them apart, which is why permanentErrorCodes exists.', 'name': 'OpenAI',
         'base': 'https://api.openai.com/v1',
         'model': 'gpt-4o-mini',
         'keys': 'platform.openai.com/api-keys',
         'note': 'Also reaches anything else served behind an OpenAI-compatible address - point the base URL at it.',
     },
     {
-        'id': 'openrouter', 'name': 'OpenRouter',
+        'id': 'openrouter',
+        'permanent_statuses': [400, 401, 402, 403],
+        'permanent_error_codes': [],
+        'retry_note': "402 is OpenRouter's own: insufficient credits. 403 covers moderation and guardrail blocks, which the same request will always trip. 408/502/503 are transient - it may reroute to another provider.", 'name': 'OpenRouter',
         'base': 'https://openrouter.ai/api/v1',
         'model': 'openai/gpt-4o-mini',
         'keys': 'openrouter.ai/keys',
         'note': 'One key for models from many providers. Model names carry the provider, as in anthropic/claude-3.5-sonnet.',
     },
     {
-        'id': 'together', 'name': 'Together AI',
+        'id': 'together',
+        'permanent_statuses': [400, 401, 402, 403, 404],
+        'permanent_error_codes': [],
+        'retry_note': "402 is the monthly spending limit. 403 is NOT a permissions error here - it means the prompt exceeded the model's context length, which retrying cannot shorten.", 'name': 'Together AI',
         'base': 'https://api.together.xyz/v1',
         'model': 'meta-llama/Llama-3.3-70B-Instruct-Turbo',
         'keys': 'api.together.ai/settings/api-keys',
         'note': 'Open-weight models, hosted.',
     },
     {
-        'id': 'groq', 'name': 'Groq',
+        'id': 'groq',
+        'permanent_statuses': [400, 401, 403, 404, 413, 499],
+        'permanent_error_codes': [],
+        'retry_note': '413 (payload too large) and 499 (caller cancelled) are permanent. 422 is deliberately absent: Groq documents it as worth retrying, because it covers semantic errors and model hallucinations. 498 is flex-tier capacity and explicitly says to try again later.', 'name': 'Groq',
         'base': 'https://api.groq.com/openai/v1',
         'model': 'llama-3.3-70b-versatile',
         'keys': 'console.groq.com/keys',
         'note': 'Very fast, a small catalogue.',
     },
     {
-        'id': 'deepseek', 'name': 'DeepSeek',
+        'id': 'deepseek',
+        'permanent_statuses': [400, 401, 402, 422],
+        'permanent_error_codes': [],
+        'retry_note': '402 is Insufficient Balance. 422 is Invalid Parameters and permanent here - note Groq treats the same status as retryable.', 'name': 'DeepSeek',
         'base': 'https://api.deepseek.com/v1',
         'model': 'deepseek-chat',
         'keys': 'platform.deepseek.com/api_keys',
         'note': 'deepseek-chat for general work, deepseek-reasoner when the answer needs working out.',
     },
     {
-        'id': 'mistral', 'name': 'Mistral',
+        'id': 'mistral',
+        'permanent_statuses': [400, 401, 403, 404, 422],
+        'permanent_error_codes': [],
+        'retry_note': "422 is a validation error and permanent - the opposite of Groq's reading of the same code.", 'name': 'Mistral',
         'base': 'https://api.mistral.ai/v1',
         'model': 'mistral-large-latest',
         'keys': 'console.mistral.ai/api-keys',
         'note': '',
     },
     {
-        'id': 'xai', 'name': 'xAI',
+        'id': 'xai',
+        'permanent_statuses': [400, 401, 403, 404],
+        'permanent_error_codes': [],
+        'retry_note': 'xAI publishes no error-code table. This is the conservative common set every OpenAI-compatible provider agrees on; nothing provider-specific is assumed.', 'name': 'xAI',
         'base': 'https://api.x.ai/v1',
         'model': 'grok-2-latest',
         'keys': 'console.x.ai',
         'note': 'The Grok models.',
     },
     {
-        'id': 'fireworks', 'name': 'Fireworks AI',
+        'id': 'fireworks',
+        'permanent_statuses': [400, 401, 402, 403, 404, 500],
+        'permanent_error_codes': [],
+        'retry_note': '500 is permanent here, alone among these providers: Fireworks documents it as a server-side code bug unlikely to resolve on its own. 402 is a billing stop. 502 and 503 stay retryable.', 'name': 'Fireworks AI',
         'base': 'https://api.fireworks.ai/inference/v1',
         'model': 'accounts/fireworks/models/llama-v3p3-70b-instruct',
         'keys': 'fireworks.ai/account/api-keys',
         'note': 'Model names are full account paths.',
     },
     {
-        'id': 'perplexity', 'name': 'Perplexity',
+        'id': 'perplexity',
+        'permanent_statuses': [400, 401, 403],
+        'permanent_error_codes': [],
+        'retry_note': '401 does double duty: an invalid key AND an account out of credits both return it, so a 401 here is not necessarily a wrong key.', 'name': 'Perplexity',
         'base': 'https://api.perplexity.ai',
         'model': 'sonar',
         'keys': 'perplexity.ai/settings/api',
@@ -83,7 +110,10 @@ PROVIDERS = [
         'no_model_list': True,
     },
     {
-        'id': 'deepinfra', 'name': 'DeepInfra',
+        'id': 'deepinfra',
+        'permanent_statuses': [400, 401, 402, 403, 404],
+        'permanent_error_codes': [],
+        'retry_note': 'DeepInfra documents little beyond 429, which is engine_overloaded and transient (a rejected request is not billed). The 4xx set is the conservative common one.', 'name': 'DeepInfra',
         'base': 'https://api.deepinfra.com/v1/openai',
         'model': 'meta-llama/Llama-3.3-70B-Instruct',
         'keys': 'deepinfra.com/dash/api_keys',
@@ -264,6 +294,15 @@ const outputSchema = {{
 
 const PROVIDER = '{name}';
 
+// Statuses this provider documents as answers that will not change.
+//
+// {retry_note}
+//
+// 429 is treated as transient unless a code below says otherwise - backing off
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
+const PERMANENT_STATUSES = {permanent_statuses};
+const PERMANENT_ERROR_CODES = {permanent_error_codes};
+
 function baseOf(config) {{
     const value = String(config.baseUrl || '{base}').trim().replace(/\/+$/, '');
     if (!value) {{
@@ -313,13 +352,24 @@ function request(options) {{
 
     if (response.status < 200 || response.status >= 300) {{
         let detail = 'HTTP ' + response.status;
+        let code = '';
         if (body && body.error) {{
             detail = typeof body.error === 'string' ? body.error :
                 (body.error.message || JSON.stringify(body.error));
+            code = (body.error && body.error.code) || (body.error && body.error.type) || '';
         }} else if (typeof body === 'string' && body) {{
             detail = body.substring(0, 200).replace(/\s+/g, ' ');
         }}
-        throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+
+        const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
+        error.status = response.status;
+        // Whether asking again could ever give a different answer. Retrying a
+        // refusal does not just waste time - where the refusal is a spent
+        // allowance or a billing stop, it spends more of whatever ran out.
+        error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
+            (PERMANENT_ERROR_CODES.length > 0 && code &&
+             PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
+        throw error;
     }}
 
     return body || {{}};
@@ -571,6 +621,11 @@ async function execute(config, input, context) {{
             parsed = null;
             smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
                 ' failed: ' + lastError);
+            if (err && err.permanent) {{
+                smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
+                    err.status + '), so the remaining attempts were not made');
+                break;
+            }}
             if (attempt < attempts) {{
                 const base = Number(config.retryDelayMs) || 2000;
                 const cap = Number(config.retryMaxDelayMs) || 30000;
@@ -661,6 +716,9 @@ def main():
             note_suffix=(' ' + note) if note else '',
             model_options='' if provider.get('no_model_list') else MODEL_OPTIONS,
             list_only=NO_LIST if provider.get('no_model_list') else LIST_ONLY,
+            permanent_statuses=provider['permanent_statuses'],
+            permanent_error_codes=provider['permanent_error_codes'],
+            retry_note=provider['retry_note'],
         )
         path = root / f"{provider['id']}-chat.js"
         path.write_text(source)

+ 25 - 0
scripts/verify-node.py

@@ -213,6 +213,12 @@ def precondition_unmet(case):
     return None
 
 
+def project_id(token):
+    """The project new rows belong to. The first one, which is what every other
+    part of the harness implicitly uses."""
+    return call("GET", "/projects", token)["projects"][0]["_id"]
+
+
 def main():
     if len(sys.argv) != 2:
         raise SystemExit("usage: verify-node.py <case.json>")
@@ -249,6 +255,23 @@ def main():
         # Substituted everywhere, so the case refers to it by name.
         case = json.loads(json.dumps(case).replace("{{helper:" + helper["key"] + "}}", helper_id))
 
+    # Credentials the case needs. Created here and deleted afterwards, so a case
+    # is portable: sdcpp-model-load-assert names a credential by id, which ties
+    # it to one installation and fails everywhere else.
+    credential_ids = []
+    for cred in case.get("credentials", []):
+        made = call("POST", "/credentials", token, {
+            "name": cred.get("name", "zz test: " + cred["key"]),
+            "type": cred["type"],
+            "projectId": project_id(token),
+            "data": cred.get("data", {}),
+        })
+        cred_id = made.get("id") or made.get("_id")
+        if not cred_id:
+            raise SystemExit(f"no id for credential {cred['key']}")
+        credential_ids.append(cred_id)
+        case = json.loads(json.dumps(case).replace("{{credential:" + cred["key"] + "}}", cred_id))
+
     created = call("POST", "/workflows", token, {
         "name": case["name"],
         "nodes": case["nodes"],
@@ -384,6 +407,8 @@ def main():
         call("DELETE", f"/workflows/{workflow_id}", token)
         for helper_id in helper_ids:
             call("DELETE", f"/workflows/{helper_id}", token)
+        for cred_id in credential_ids:
+            call("DELETE", f"/credentials/{cred_id}", token)
 
 
 if __name__ == "__main__":

+ 28 - 0
tests/nodes/chat-permanent-status-stops-retrying.json

@@ -0,0 +1,28 @@
+{
+  "name": "chat-permanent-status-stops-retrying",
+  "description": "Groq documents 404 as permanent, so the node must stop at the first refusal instead of spending its retries. Pointed at this API's own /chat/completions, which does not exist and answers 404. With retryCount 2 a retrying node reports attempts 3; stopping reports 1.",
+  "credentials": [
+    {"key": "fake", "name": "zz test: groq permanent status", "type": "api_key",
+     "data": {"keyValue": "not-a-real-key", "headerName": "Authorization", "valuePrefix": "Bearer "}}
+  ],
+  "nodes": [
+    {"id": "n1", "name": "Trigger", "type": "click-trigger", "position": {"x": 0, "y": 0}, "config": {}},
+    {"id": "n2", "name": "Refused", "type": "groq-chat", "position": {"x": 0, "y": 100},
+     "config": {
+       "credentialId": "{{credential:fake}}",
+       "baseUrl": "{{API}}/api/v1",
+       "model": "llama-3.3-70b-versatile",
+       "userPrompt": "hello",
+       "retryCount": 2,
+       "retryDelayMs": 100,
+       "skipOnError": true,
+       "timeoutMs": 15000
+     }}
+  ],
+  "connections": [
+    {"sourceNodeId": "n1", "sourceOutput": "main", "targetNodeId": "n2", "targetInput": "data"}
+  ],
+  "expect": {
+    "n2": {"status": "completed", "output": {"success": false, "attempts": 1}}
+  }
+}

+ 64 - 0
tests/nodes/chat-transient-status-still-retries.json

@@ -0,0 +1,64 @@
+{
+  "name": "chat-transient-status-still-retries",
+  "description": "The same 404 from the same endpoint, through OpenAI's node. OpenAI's error documentation lists 400, 401 and 403 as the refusals - not 404 - so this one keeps its retries and reports attempts 3. The pair is the point: the status is identical and the correct behaviour differs, which is why the policy is per provider and not shared.",
+  "credentials": [
+    {
+      "key": "fake",
+      "name": "zz test: openai transient status",
+      "type": "api_key",
+      "data": {
+        "keyValue": "not-a-real-key",
+        "headerName": "Authorization",
+        "valuePrefix": "Bearer "
+      }
+    }
+  ],
+  "nodes": [
+    {
+      "id": "n1",
+      "name": "Trigger",
+      "type": "click-trigger",
+      "position": {
+        "x": 0,
+        "y": 0
+      },
+      "config": {}
+    },
+    {
+      "id": "n2",
+      "name": "Refused",
+      "type": "openai-chat",
+      "position": {
+        "x": 0,
+        "y": 100
+      },
+      "config": {
+        "credentialId": "{{credential:fake}}",
+        "baseUrl": "{{API}}/api/v1",
+        "model": "gpt-4o-mini",
+        "userPrompt": "hello",
+        "retryCount": 2,
+        "retryDelayMs": 100,
+        "skipOnError": true,
+        "timeoutMs": 15000
+      }
+    }
+  ],
+  "connections": [
+    {
+      "sourceNodeId": "n1",
+      "sourceOutput": "main",
+      "targetNodeId": "n2",
+      "targetInput": "data"
+    }
+  ],
+  "expect": {
+    "n2": {
+      "status": "completed",
+      "output": {
+        "success": false,
+        "attempts": 3
+      }
+    }
+  }
+}