Repository navigation
fix(llm): show known provider failures as plain sentences #282
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
0f6d81d
7005f13
8e7752d
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,107 @@ | ||
| """Plain sentences for the ways a model provider says no (issue #280). | ||
|
|
||
| LiteLLM's errors carry the provider's raw reply (``litellm.RateLimitError: ... OpenrouterException - {"error": ...``). | ||
| ``plain_error`` turns the failures people actually hit into one sentence that says what to do. The raw text only | ||
| goes to the log; anything unknown returns None and keeps its own message. | ||
| """ | ||
|
|
||
| from __future__ import annotations | ||
|
|
||
| SETTINGS = "Settings > Models" | ||
|
|
||
| LABELS = { | ||
| "openrouter": "OpenRouter", "anthropic": "Anthropic", "openai": "OpenAI", "gemini": "Google Gemini", | ||
| "vertex_ai": "Google Vertex AI", "groq": "Groq", "mistral": "Mistral", "deepseek": "DeepSeek", "xai": "xAI", | ||
| "together_ai": "Together AI", "fireworks_ai": "Fireworks AI", "cohere": "Cohere", "perplexity": "Perplexity", | ||
| "nous": "Nous Portal", "azure": "Azure OpenAI", "bedrock": "Amazon Bedrock", "ollama": "Ollama", | ||
| "ollama_chat": "Ollama", "lm_studio": "LM Studio", "chatgpt": "ChatGPT", | ||
| } | ||
| # where each provider sells credits, for "out of credits" | ||
| BILLING = { | ||
| "openrouter": "openrouter.ai/settings/credits", "anthropic": "console.anthropic.com", | ||
| "openai": "platform.openai.com", "deepseek": "platform.deepseek.com", "xai": "console.x.ai", | ||
| "mistral": "console.mistral.ai", "nous": "portal.nousresearch.com", | ||
| } | ||
| LOCAL = {"ollama", "ollama_chat", "lm_studio", "llamafile", "vllm", "hosted_vllm"} | ||
|
|
||
| CREDIT_HINTS = ( | ||
| "requires more credits", "insufficient credits", "insufficient_quota", "exceeded your current quota", | ||
| "credit balance is too low", "insufficient balance", "out of credits", "payment required", | ||
| ) | ||
| KEY_HINTS = ("invalid api key", "invalid x-api-key", "incorrect api key", "api key not valid", "invalid_api_key", | ||
| "user not found", "no auth credentials") | ||
| NOT_FOUND_HINTS = ("not a valid model id", "model_not_found", "does not exist", "not found, try pulling it first", | ||
| "unknown model", "no such model") | ||
| UNREACHABLE_HINTS = ("cannot connect to host", "all connection attempts failed", "connection refused", | ||
| "refused the network connection", "connecterror", "getaddrinfo failed", "name or service not known", "nodename nor servname") | ||
|
|
||
|
|
||
| def _parts(model: str) -> tuple[str, str]: | ||
| prefix, _, name = model.partition("/") | ||
| return (prefix, name) if name else ("", model) | ||
|
|
||
|
|
||
| def _label(prefix: str) -> str: | ||
| return LABELS.get(prefix) or (prefix.replace("_", " ").title() if prefix else "The model's provider") | ||
|
|
||
|
|
||
| def _status(exc: BaseException) -> int | None: | ||
| code = getattr(exc, "status_code", None) | ||
| return code if isinstance(code, int) else None | ||
|
|
||
|
|
||
| def is_tool_support_error(text: str) -> bool: | ||
| """The provider turned the request down because the model can't call tools (the agent loop then answers | ||
| without tools, so these keep their own message).""" | ||
| m = text.lower() | ||
| return ("does not support tools" in m or "support tool use" in m or "invalid character" in m | ||
| or ("tool" in m and "pars" in m)) | ||
|
|
||
|
|
||
| def plain_error(exc: BaseException, model: str) -> str | None: | ||
| """A plain sentence for a known provider failure of ``model`` that says what to do, or None when unknown. | ||
|
|
||
| Only LiteLLM's own errors are translated: Sentient's errors (Claude Code, the ChatGPT plan) are already plain. | ||
| """ | ||
| if not type(exc).__module__.startswith("litellm"): | ||
| return None | ||
| text = str(exc) | ||
| m = text.lower() | ||
| if is_tool_support_error(m): | ||
| return None | ||
| prefix, name = _parts(model) | ||
| label, status, kind = _label(prefix), _status(exc), type(exc).__name__ | ||
| pick = f"pick another model in {SETTINGS}" | ||
|
|
||
| if "agentic harness" in m: | ||
| return (f"{label} only lets coding tools use {name}, not assistants like Sentient. " | ||
| f"Please {pick}.") | ||
| if "free-models-per-day" in m: | ||
| return (f"{label}'s free models have reached today's limit. Add credits on openrouter.ai, wait until " | ||
| f"tomorrow, or {pick}.") | ||
| if "free-models-per-min" in m: | ||
| return f"{label}'s free models allow only a few requests a minute. Wait a minute and try again, or {pick}." | ||
| if status == 402 or any(h in m for h in CREDIT_HINTS): | ||
| where = f" on {BILLING[prefix]}" if prefix in BILLING else "" | ||
| return f"Your {label} account is out of credits. Add credits{where}, or {pick}." | ||
| if "no endpoints found matching your data policy" in m: | ||
| return (f"Your {label} privacy settings don't allow any provider of {name}. Change them on " | ||
| f"openrouter.ai/settings/privacy, or {pick}.") | ||
| if status == 401 or kind == "AuthenticationError" or any(h in m for h in KEY_HINTS): | ||
| return f"{label} didn't accept your key. It may be mistyped or revoked: add a new one in {SETTINGS}." | ||
| if status == 404 or kind == "NotFoundError" or any(h in m for h in NOT_FOUND_HINTS): | ||
| if prefix in {"ollama", "ollama_chat"}: | ||
| return f"{name} isn't downloaded in Ollama. Download it in {SETTINGS}, or pick another model there." | ||
| return f"{label} doesn't have a model called {name}. Check the name, or {pick}." | ||
| if status == 429 or kind == "RateLimitError": | ||
| return f"{label} is getting too many requests right now. Wait a minute and try again, or {pick}." | ||
| if status == 408 or kind == "Timeout": | ||
| return f"{label} took too long to answer. Try again in a moment, or {pick}." | ||
| if any(h in m for h in UNREACHABLE_HINTS): | ||
| if prefix in LOCAL: | ||
| return f"Sentient couldn't reach {label}. Make sure it's running, then try again." | ||
| return f"Sentient couldn't reach {label}. Check your internet connection and try again." | ||
| if (kind in {"InternalServerError", "ServiceUnavailableError", "BadGatewayError"} | ||
| or (kind != "APIConnectionError" and status is not None and status >= 500) or "overloaded" in m): | ||
| return f"{label} is having trouble right now. Try again in a few minutes, or {pick}." | ||
| return None |
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -20,7 +20,10 @@ | |
|
|
||
| from sentient import secrets | ||
| from sentient.config.schema import ProviderConfig, SentientConfig | ||
| from sentient.llm.chatgpt import ChatGPTError | ||
| from sentient.llm.errors import is_tool_support_error, plain_error | ||
| from sentient.llm.jobs import ModelJobs | ||
| from sentient.llm.responses import ResponsesError | ||
|
|
||
| log = logging.getLogger(__name__) | ||
|
|
||
|
|
@@ -29,15 +32,39 @@ class ProviderError(RuntimeError): | |
| pass | ||
|
|
||
|
|
||
| class ModelRefused(ProviderError): | ||
| class PlainProviderError(ProviderError): | ||
| """The message is a plain sentence that says what to do, for the user as is (the raw error is only logged).""" | ||
|
|
||
|
|
||
| class ModelRefused(PlainProviderError): | ||
| """A model that is set up can't do this job (Claude Code in the background, for one). The message is a plain | ||
| sentence that says what to change, for the user as is.""" | ||
|
|
||
|
|
||
| class ProviderFailed(PlainProviderError): | ||
| """A provider turned the request down for a known reason: out of credits, a rate limit, a bad key... (#280).""" | ||
|
|
||
|
|
||
| def _failure(exc: Exception, model: str) -> Exception: | ||
| """What a failed model leaves for the user: a plain error for a known provider failure, else ``exc`` itself. | ||
| Claude Code and ChatGPT plan errors are written for the user already.""" | ||
| if isinstance(exc, PlainProviderError): | ||
| return exc | ||
| if (isinstance(exc, ResponsesError | ChatGPTError) or ( | ||
| _provider_prefix(model) == CLAUDE_CODE and type(exc) is ProviderError | ||
| )) and not is_tool_support_error(str(exc)): # a tool refusal stays as is: chat answers without tools | ||
| return ProviderFailed(str(exc)) | ||
| plain = plain_error(exc, model) | ||
| return ProviderFailed(plain) if plain else exc | ||
|
Comment on lines
+48
to
+58
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🔒 Security & Privacy | 🛡️ Detected with Advanced Tier | 🟡 Minor | ⚡ Quick win 🧩 Analysis chain🏁 Script executed: #!/bin/bash
rg -n -C6 '_problem|raw\[:200\]' sentient/llm/responses.pyRepository: existence-master/Sentient Length of output: 1560 🏁 Script executed: #!/bin/bash
set -e
printf '%s\n' '--- provider.py _failure and surrounding definitions ---'
sed -n '25,75p' sentient/llm/provider.py
printf '%s\n' '--- responses.py error formatting and request failure ---'
sed -n '120,195p' sentient/llm/responses.py
printf '%s\n' '--- ChatGPTError definition and raise sites ---'
rg -n -C4 --glob 'sentient/llm/chatgpt.py' 'ChatGPTError|raise .*Error' sentient/llm/chatgpt.py || test "$?" -eq 1
printf '%s\n' '--- Claude Code ProviderError raise sites ---'
rg -n -C4 --glob 'sentient/llm/claude_code.py' 'ProviderError|raise .*Error' sentient/llm/claude_code.py || test "$?" -eq 1Repository: existence-master/Sentient Length of output: 19599 🏁 Script executed: #!/bin/bash
set -e
printf '%s\n' '--- ChatGPT error helpers ---'
sed -n '90,145p' sentient/llm/chatgpt.py
printf '%s\n' '--- ChatGPT model-list error path ---'
sed -n '328,347p' sentient/llm/chatgpt.py
printf '%s\n' '--- Responses problem formatter ---'
sed -n '80,125p' sentient/llm/responses.py
printf '%s\n' '--- Claude Code result error helper and constants ---'
rg -n -C8 --glob 'sentient/llm/claude_code.py' '_result_error|STDERR_KEEP|is_error' sentient/llm/claude_code.py || test "$?" -eq 1Repository: existence-master/Sentient Length of output: 10497 Information Disclosure Reachability: External Do not expose provider-controlled error details through 🤖 Prompt for AI AgentsSource: Learnings |
||
|
|
||
|
|
||
| def _all_failed(role: str, errors: list[Exception]) -> ProviderError: | ||
| """The error for a role whose every model failed. When every model refused the job, say why and what to do.""" | ||
| """The error for a role whose every model failed: the last model's plain reason when Sentient knows it (when | ||
| every model refused the job, why and what to do), else the general error.""" | ||
| if errors and all(isinstance(e, ModelRefused) for e in errors): | ||
| return ModelRefused(str(errors[0])) | ||
| if errors and isinstance(errors[-1], PlainProviderError): | ||
| return ProviderFailed(str(errors[-1])) | ||
| return ProviderError(f"All models failed for role '{role}': {errors[-1] if errors else None}") | ||
|
|
||
|
|
||
|
|
@@ -348,10 +375,13 @@ async def stream( | |
| ) | ||
| return | ||
| except Exception as exc: | ||
| errors.append(exc) | ||
| failure = _failure(exc, model) | ||
| errors.append(failure) | ||
| log.warning("model %s failed for role %s: %s", model, role, exc) | ||
| if emitted: | ||
| # part of a reply already reached the user; switching models would duplicate it | ||
| if isinstance(failure, PlainProviderError): | ||
| raise ProviderFailed(str(failure)) from exc | ||
| raise ProviderError(f"{model} stopped mid-reply: {exc}") from exc | ||
| continue | ||
| raise _all_failed(role, errors) | ||
|
|
@@ -414,7 +444,7 @@ async def complete_text(self, role: str, messages: list[dict], *, model: str | N | |
| text = resp.choices[0].message.content or "" | ||
| return re.sub(r"<think>.*?</think>", "", text, flags=re.DOTALL).strip() | ||
| except Exception as exc: | ||
| errors.append(exc) | ||
| errors.append(_failure(exc, model)) | ||
| log.warning("model %s failed for role %s: %s", model, role, exc) | ||
| raise _all_failed(role, errors) | ||
|
|
||
|
|
@@ -439,7 +469,7 @@ async def complete_json(self, role: str, messages: list[dict], *, model: str | N | |
| text = resp.choices[0].message.content or "" | ||
| return parse_json_loose(text) | ||
| except Exception as exc: | ||
| errors.append(exc) | ||
| errors.append(_failure(exc, model)) | ||
| log.warning("model %s failed for role %s: %s", model, role, exc) | ||
| raise _all_failed(role, errors) | ||
|
|
||
|
|
@@ -456,8 +486,15 @@ async def embed(self, texts: list[str], *, model: str | None = None) -> list[lis | |
| raise ProviderError("ChatGPT plans don't include embedding models. Pick a local or API embedding model.") | ||
| kwargs = self._kwargs_for(model) | ||
| kwargs.pop("timeout", None) | ||
| async with self.jobs.slot(model): | ||
| resp = await litellm.aembedding(model=litellm_model(model), input=texts, **kwargs) | ||
| try: | ||
| async with self.jobs.slot(model): | ||
| resp = await litellm.aembedding(model=litellm_model(model), input=texts, **kwargs) | ||
| except Exception as exc: | ||
| plain = plain_error(exc, model) | ||
| if plain is None: | ||
| raise | ||
| log.warning("embedding model %s failed: %s", model, exc) | ||
| raise ProviderFailed(plain) from exc | ||
| return [d["embedding"] for d in resp.data] | ||
|
|
||
|
|
||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
🔒 Security & Privacy | 🟡 Minor | ⚡ Quick win
🔎 Supported by static analysis
🏁 Script executed:
Repository: existence-master/Sentient
Length of output: 42432
🏁 Script executed:
Repository: existence-master/Sentient
Length of output: 42345
🏁 Script executed:
Repository: existence-master/Sentient
Length of output: 20039
Reachability path
Keep upstream response bodies out of
ResponsesErrormessages.A non-JSON HTTP error passes the first 200 bytes of the upstream body to
ResponsesError._failurethen turns that text intoProviderFailed, and chat and task paths expose it to users. This violates the contract that raw provider replies go only to logs.Suggested fix
def _problem(status: int, raw: bytes) -> str: try: data = json.loads(raw) except ValueError: - return problem_text(status, "", raw[:200].decode("utf-8", "replace")) - code, detail = _error_fields(data.get("error") if isinstance(data, dict) else data) - return problem_text(status, code, detail) + return problem_text(status, "", "") + code, _ = _error_fields(data.get("error") if isinstance(data, dict) else data) + return problem_text(status, code, "")🤖 Prompt for AI Agents