|
@@ -19,63 +19,90 @@ from pathlib import Path
|
|
|
|
|
|
|
|
PROVIDERS = [
|
|
PROVIDERS = [
|
|
|
{
|
|
{
|
|
|
- 'id': 'openai', 'name': 'OpenAI',
|
|
|
|
|
|
|
+ 'id': 'openai',
|
|
|
|
|
+ 'permanent_statuses': [400, 401, 403],
|
|
|
|
|
+ 'permanent_error_codes': ['insufficient_quota', 'billing_hard_limit_reached', 'access_terminated'],
|
|
|
|
|
+ 'retry_note': 'OpenAI splits 429: "rate limit reached" is transient, but an exhausted credit balance, a project spend limit and an organisation usage limit all arrive as 429 too and no amount of backing off clears them. The error code tells them apart, which is why permanentErrorCodes exists.', 'name': 'OpenAI',
|
|
|
'base': 'https://api.openai.com/v1',
|
|
'base': 'https://api.openai.com/v1',
|
|
|
'model': 'gpt-4o-mini',
|
|
'model': 'gpt-4o-mini',
|
|
|
'keys': 'platform.openai.com/api-keys',
|
|
'keys': 'platform.openai.com/api-keys',
|
|
|
'note': 'Also reaches anything else served behind an OpenAI-compatible address - point the base URL at it.',
|
|
'note': 'Also reaches anything else served behind an OpenAI-compatible address - point the base URL at it.',
|
|
|
},
|
|
},
|
|
|
{
|
|
{
|
|
|
- 'id': 'openrouter', 'name': 'OpenRouter',
|
|
|
|
|
|
|
+ 'id': 'openrouter',
|
|
|
|
|
+ 'permanent_statuses': [400, 401, 402, 403],
|
|
|
|
|
+ 'permanent_error_codes': [],
|
|
|
|
|
+ 'retry_note': "402 is OpenRouter's own: insufficient credits. 403 covers moderation and guardrail blocks, which the same request will always trip. 408/502/503 are transient - it may reroute to another provider.", 'name': 'OpenRouter',
|
|
|
'base': 'https://openrouter.ai/api/v1',
|
|
'base': 'https://openrouter.ai/api/v1',
|
|
|
'model': 'openai/gpt-4o-mini',
|
|
'model': 'openai/gpt-4o-mini',
|
|
|
'keys': 'openrouter.ai/keys',
|
|
'keys': 'openrouter.ai/keys',
|
|
|
'note': 'One key for models from many providers. Model names carry the provider, as in anthropic/claude-3.5-sonnet.',
|
|
'note': 'One key for models from many providers. Model names carry the provider, as in anthropic/claude-3.5-sonnet.',
|
|
|
},
|
|
},
|
|
|
{
|
|
{
|
|
|
- 'id': 'together', 'name': 'Together AI',
|
|
|
|
|
|
|
+ 'id': 'together',
|
|
|
|
|
+ 'permanent_statuses': [400, 401, 402, 403, 404],
|
|
|
|
|
+ 'permanent_error_codes': [],
|
|
|
|
|
+ 'retry_note': "402 is the monthly spending limit. 403 is NOT a permissions error here - it means the prompt exceeded the model's context length, which retrying cannot shorten.", 'name': 'Together AI',
|
|
|
'base': 'https://api.together.xyz/v1',
|
|
'base': 'https://api.together.xyz/v1',
|
|
|
'model': 'meta-llama/Llama-3.3-70B-Instruct-Turbo',
|
|
'model': 'meta-llama/Llama-3.3-70B-Instruct-Turbo',
|
|
|
'keys': 'api.together.ai/settings/api-keys',
|
|
'keys': 'api.together.ai/settings/api-keys',
|
|
|
'note': 'Open-weight models, hosted.',
|
|
'note': 'Open-weight models, hosted.',
|
|
|
},
|
|
},
|
|
|
{
|
|
{
|
|
|
- 'id': 'groq', 'name': 'Groq',
|
|
|
|
|
|
|
+ 'id': 'groq',
|
|
|
|
|
+ 'permanent_statuses': [400, 401, 403, 404, 413, 499],
|
|
|
|
|
+ 'permanent_error_codes': [],
|
|
|
|
|
+ 'retry_note': '413 (payload too large) and 499 (caller cancelled) are permanent. 422 is deliberately absent: Groq documents it as worth retrying, because it covers semantic errors and model hallucinations. 498 is flex-tier capacity and explicitly says to try again later.', 'name': 'Groq',
|
|
|
'base': 'https://api.groq.com/openai/v1',
|
|
'base': 'https://api.groq.com/openai/v1',
|
|
|
'model': 'llama-3.3-70b-versatile',
|
|
'model': 'llama-3.3-70b-versatile',
|
|
|
'keys': 'console.groq.com/keys',
|
|
'keys': 'console.groq.com/keys',
|
|
|
'note': 'Very fast, a small catalogue.',
|
|
'note': 'Very fast, a small catalogue.',
|
|
|
},
|
|
},
|
|
|
{
|
|
{
|
|
|
- 'id': 'deepseek', 'name': 'DeepSeek',
|
|
|
|
|
|
|
+ 'id': 'deepseek',
|
|
|
|
|
+ 'permanent_statuses': [400, 401, 402, 422],
|
|
|
|
|
+ 'permanent_error_codes': [],
|
|
|
|
|
+ 'retry_note': '402 is Insufficient Balance. 422 is Invalid Parameters and permanent here - note Groq treats the same status as retryable.', 'name': 'DeepSeek',
|
|
|
'base': 'https://api.deepseek.com/v1',
|
|
'base': 'https://api.deepseek.com/v1',
|
|
|
'model': 'deepseek-chat',
|
|
'model': 'deepseek-chat',
|
|
|
'keys': 'platform.deepseek.com/api_keys',
|
|
'keys': 'platform.deepseek.com/api_keys',
|
|
|
'note': 'deepseek-chat for general work, deepseek-reasoner when the answer needs working out.',
|
|
'note': 'deepseek-chat for general work, deepseek-reasoner when the answer needs working out.',
|
|
|
},
|
|
},
|
|
|
{
|
|
{
|
|
|
- 'id': 'mistral', 'name': 'Mistral',
|
|
|
|
|
|
|
+ 'id': 'mistral',
|
|
|
|
|
+ 'permanent_statuses': [400, 401, 403, 404, 422],
|
|
|
|
|
+ 'permanent_error_codes': [],
|
|
|
|
|
+ 'retry_note': "422 is a validation error and permanent - the opposite of Groq's reading of the same code.", 'name': 'Mistral',
|
|
|
'base': 'https://api.mistral.ai/v1',
|
|
'base': 'https://api.mistral.ai/v1',
|
|
|
'model': 'mistral-large-latest',
|
|
'model': 'mistral-large-latest',
|
|
|
'keys': 'console.mistral.ai/api-keys',
|
|
'keys': 'console.mistral.ai/api-keys',
|
|
|
'note': '',
|
|
'note': '',
|
|
|
},
|
|
},
|
|
|
{
|
|
{
|
|
|
- 'id': 'xai', 'name': 'xAI',
|
|
|
|
|
|
|
+ 'id': 'xai',
|
|
|
|
|
+ 'permanent_statuses': [400, 401, 403, 404],
|
|
|
|
|
+ 'permanent_error_codes': [],
|
|
|
|
|
+ 'retry_note': 'xAI publishes no error-code table. This is the conservative common set every OpenAI-compatible provider agrees on; nothing provider-specific is assumed.', 'name': 'xAI',
|
|
|
'base': 'https://api.x.ai/v1',
|
|
'base': 'https://api.x.ai/v1',
|
|
|
'model': 'grok-2-latest',
|
|
'model': 'grok-2-latest',
|
|
|
'keys': 'console.x.ai',
|
|
'keys': 'console.x.ai',
|
|
|
'note': 'The Grok models.',
|
|
'note': 'The Grok models.',
|
|
|
},
|
|
},
|
|
|
{
|
|
{
|
|
|
- 'id': 'fireworks', 'name': 'Fireworks AI',
|
|
|
|
|
|
|
+ 'id': 'fireworks',
|
|
|
|
|
+ 'permanent_statuses': [400, 401, 402, 403, 404, 500],
|
|
|
|
|
+ 'permanent_error_codes': [],
|
|
|
|
|
+ 'retry_note': '500 is permanent here, alone among these providers: Fireworks documents it as a server-side code bug unlikely to resolve on its own. 402 is a billing stop. 502 and 503 stay retryable.', 'name': 'Fireworks AI',
|
|
|
'base': 'https://api.fireworks.ai/inference/v1',
|
|
'base': 'https://api.fireworks.ai/inference/v1',
|
|
|
'model': 'accounts/fireworks/models/llama-v3p3-70b-instruct',
|
|
'model': 'accounts/fireworks/models/llama-v3p3-70b-instruct',
|
|
|
'keys': 'fireworks.ai/account/api-keys',
|
|
'keys': 'fireworks.ai/account/api-keys',
|
|
|
'note': 'Model names are full account paths.',
|
|
'note': 'Model names are full account paths.',
|
|
|
},
|
|
},
|
|
|
{
|
|
{
|
|
|
- 'id': 'perplexity', 'name': 'Perplexity',
|
|
|
|
|
|
|
+ 'id': 'perplexity',
|
|
|
|
|
+ 'permanent_statuses': [400, 401, 403],
|
|
|
|
|
+ 'permanent_error_codes': [],
|
|
|
|
|
+ 'retry_note': '401 does double duty: an invalid key AND an account out of credits both return it, so a 401 here is not necessarily a wrong key.', 'name': 'Perplexity',
|
|
|
'base': 'https://api.perplexity.ai',
|
|
'base': 'https://api.perplexity.ai',
|
|
|
'model': 'sonar',
|
|
'model': 'sonar',
|
|
|
'keys': 'perplexity.ai/settings/api',
|
|
'keys': 'perplexity.ai/settings/api',
|
|
@@ -83,7 +110,10 @@ PROVIDERS = [
|
|
|
'no_model_list': True,
|
|
'no_model_list': True,
|
|
|
},
|
|
},
|
|
|
{
|
|
{
|
|
|
- 'id': 'deepinfra', 'name': 'DeepInfra',
|
|
|
|
|
|
|
+ 'id': 'deepinfra',
|
|
|
|
|
+ 'permanent_statuses': [400, 401, 402, 403, 404],
|
|
|
|
|
+ 'permanent_error_codes': [],
|
|
|
|
|
+ 'retry_note': 'DeepInfra documents little beyond 429, which is engine_overloaded and transient (a rejected request is not billed). The 4xx set is the conservative common one.', 'name': 'DeepInfra',
|
|
|
'base': 'https://api.deepinfra.com/v1/openai',
|
|
'base': 'https://api.deepinfra.com/v1/openai',
|
|
|
'model': 'meta-llama/Llama-3.3-70B-Instruct',
|
|
'model': 'meta-llama/Llama-3.3-70B-Instruct',
|
|
|
'keys': 'deepinfra.com/dash/api_keys',
|
|
'keys': 'deepinfra.com/dash/api_keys',
|
|
@@ -264,6 +294,15 @@ const outputSchema = {{
|
|
|
|
|
|
|
|
const PROVIDER = '{name}';
|
|
const PROVIDER = '{name}';
|
|
|
|
|
|
|
|
|
|
+// Statuses this provider documents as answers that will not change.
|
|
|
|
|
+//
|
|
|
|
|
+// {retry_note}
|
|
|
|
|
+//
|
|
|
|
|
+// 429 is treated as transient unless a code below says otherwise - backing off
|
|
|
|
|
+// is exactly what a rate limit asks for. 5xx is transient unless listed.
|
|
|
|
|
+const PERMANENT_STATUSES = {permanent_statuses};
|
|
|
|
|
+const PERMANENT_ERROR_CODES = {permanent_error_codes};
|
|
|
|
|
+
|
|
|
function baseOf(config) {{
|
|
function baseOf(config) {{
|
|
|
const value = String(config.baseUrl || '{base}').trim().replace(/\/+$/, '');
|
|
const value = String(config.baseUrl || '{base}').trim().replace(/\/+$/, '');
|
|
|
if (!value) {{
|
|
if (!value) {{
|
|
@@ -313,13 +352,24 @@ function request(options) {{
|
|
|
|
|
|
|
|
if (response.status < 200 || response.status >= 300) {{
|
|
if (response.status < 200 || response.status >= 300) {{
|
|
|
let detail = 'HTTP ' + response.status;
|
|
let detail = 'HTTP ' + response.status;
|
|
|
|
|
+ let code = '';
|
|
|
if (body && body.error) {{
|
|
if (body && body.error) {{
|
|
|
detail = typeof body.error === 'string' ? body.error :
|
|
detail = typeof body.error === 'string' ? body.error :
|
|
|
(body.error.message || JSON.stringify(body.error));
|
|
(body.error.message || JSON.stringify(body.error));
|
|
|
|
|
+ code = (body.error && body.error.code) || (body.error && body.error.type) || '';
|
|
|
}} else if (typeof body === 'string' && body) {{
|
|
}} else if (typeof body === 'string' && body) {{
|
|
|
detail = body.substring(0, 200).replace(/\s+/g, ' ');
|
|
detail = body.substring(0, 200).replace(/\s+/g, ' ');
|
|
|
}}
|
|
}}
|
|
|
- throw new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
|
|
|
|
|
|
|
+
|
|
|
|
|
+ const error = new Error(PROVIDER + ' ' + options.what + ' failed: ' + detail);
|
|
|
|
|
+ error.status = response.status;
|
|
|
|
|
+ // Whether asking again could ever give a different answer. Retrying a
|
|
|
|
|
+ // refusal does not just waste time - where the refusal is a spent
|
|
|
|
|
+ // allowance or a billing stop, it spends more of whatever ran out.
|
|
|
|
|
+ error.permanent = PERMANENT_STATUSES.indexOf(response.status) !== -1 ||
|
|
|
|
|
+ (PERMANENT_ERROR_CODES.length > 0 && code &&
|
|
|
|
|
+ PERMANENT_ERROR_CODES.indexOf(String(code)) !== -1);
|
|
|
|
|
+ throw error;
|
|
|
}}
|
|
}}
|
|
|
|
|
|
|
|
return body || {{}};
|
|
return body || {{}};
|
|
@@ -571,6 +621,11 @@ async function execute(config, input, context) {{
|
|
|
parsed = null;
|
|
parsed = null;
|
|
|
smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
|
|
smartbotic.log.warn(PROVIDER + ': attempt ' + attempt + ' of ' + attempts +
|
|
|
' failed: ' + lastError);
|
|
' failed: ' + lastError);
|
|
|
|
|
+ if (err && err.permanent) {{
|
|
|
|
|
+ smartbotic.log.warn(PROVIDER + ': the request was refused (HTTP ' +
|
|
|
|
|
+ err.status + '), so the remaining attempts were not made');
|
|
|
|
|
+ break;
|
|
|
|
|
+ }}
|
|
|
if (attempt < attempts) {{
|
|
if (attempt < attempts) {{
|
|
|
const base = Number(config.retryDelayMs) || 2000;
|
|
const base = Number(config.retryDelayMs) || 2000;
|
|
|
const cap = Number(config.retryMaxDelayMs) || 30000;
|
|
const cap = Number(config.retryMaxDelayMs) || 30000;
|
|
@@ -661,6 +716,9 @@ def main():
|
|
|
note_suffix=(' ' + note) if note else '',
|
|
note_suffix=(' ' + note) if note else '',
|
|
|
model_options='' if provider.get('no_model_list') else MODEL_OPTIONS,
|
|
model_options='' if provider.get('no_model_list') else MODEL_OPTIONS,
|
|
|
list_only=NO_LIST if provider.get('no_model_list') else LIST_ONLY,
|
|
list_only=NO_LIST if provider.get('no_model_list') else LIST_ONLY,
|
|
|
|
|
+ permanent_statuses=provider['permanent_statuses'],
|
|
|
|
|
+ permanent_error_codes=provider['permanent_error_codes'],
|
|
|
|
|
+ retry_note=provider['retry_note'],
|
|
|
)
|
|
)
|
|
|
path = root / f"{provider['id']}-chat.js"
|
|
path = root / f"{provider['id']}-chat.js"
|
|
|
path.write_text(source)
|
|
path.write_text(source)
|