[fix] plugin: ask the LLM server twice before giving up
An idle LLM server unloads the model and loads it again on the next request. Nothing of the response is sent while that load runs, so a request that arrives on a cold server can exceed read_timeout and fail -- reliably making the first search after an idle period the one that does not get a summary. That same request is what starts the load, so asking again succeeds. Only a server that failed to answer is asked again. A server that did answer with an error status is not: a wrong API key or an unknown model name does not become right on a second attempt. The status and body of such an error response are now logged. Until now every upstream failure looked identical from the outside -- HTTP 502 with no indication of whether the key, the model name or the network was at fault. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
ce02bebf7b
commit
4a582c0a15
+55
-11
@@ -47,6 +47,18 @@ if t.TYPE_CHECKING:
|
||||
VALID_ROLES = ("user", "assistant")
|
||||
MODEL_NAME_REGEXP = re.compile(r"[A-Za-z0-9._:/-]{1,128}")
|
||||
|
||||
log = logging.getLogger("searx.plugins.ai_summary")
|
||||
|
||||
UPSTREAM_RETRIES = 1
|
||||
"""How often a request to the LLM server is repeated when the server does not
|
||||
answer in time.
|
||||
|
||||
An idle LLM server unloads the model, and loads it again on the next request --
|
||||
which can take longer than :py:obj:`read_timeout
|
||||
<searx.ai_summary.SettingsAISummary.read_timeout>`, because no byte of the
|
||||
response is sent while the model is loading. The request that runs into this
|
||||
is also the request that starts the load, so repeating it usually succeeds."""
|
||||
|
||||
|
||||
def _get_client(base_url: str, cfg: SettingsAISummary, api_key: str = "") -> httpx.Client:
|
||||
"""HTTP client for one request to the LLM server at ``base_url``. The
|
||||
@@ -238,6 +250,47 @@ def _validate_payload(payload: t.Any, cfg: SettingsAISummary) -> tuple[list[dict
|
||||
return _validate_messages(payload.get("messages"), cfg), _validate_context(payload.get("context", []), cfg)
|
||||
|
||||
|
||||
def _open_upstream(client: httpx.Client, payload: dict[str, t.Any]) -> tuple[t.Any, t.Any]:
|
||||
"""Start the streaming completion on the LLM server.
|
||||
|
||||
Returns the (already entered) stream context and the response, or
|
||||
``(None, None)`` if no usable response was received. A server that does not
|
||||
answer in time is asked again (:py:obj:`UPSTREAM_RETRIES`); a server that
|
||||
*did* answer, but with an error status, is not -- a wrong API key or an
|
||||
unknown model name does not become right when asked twice."""
|
||||
|
||||
for attempt in range(UPSTREAM_RETRIES + 1):
|
||||
stream_ctx = client.stream("POST", "/chat/completions", json=payload)
|
||||
try:
|
||||
resp = stream_ctx.__enter__() # pylint: disable=unnecessary-dunder-call
|
||||
except httpx.TransportError as exc:
|
||||
if attempt < UPSTREAM_RETRIES:
|
||||
log.debug("LLM server did not answer (%s), asking again", exc)
|
||||
continue
|
||||
log.warning("LLM server did not answer: %s", exc)
|
||||
return None, None
|
||||
except httpx.HTTPError as exc:
|
||||
log.warning("request to the LLM server failed: %s", exc)
|
||||
return None, None
|
||||
|
||||
if resp.status_code == 200:
|
||||
return stream_ctx, resp
|
||||
|
||||
# the body of an error response is short and usually names the cause,
|
||||
# e.g. an unknown model; without it a misconfiguration is invisible
|
||||
detail = ""
|
||||
try:
|
||||
resp.read()
|
||||
detail = resp.text.strip()[:200]
|
||||
except (httpx.HTTPError, UnicodeDecodeError): # pragma: no cover
|
||||
pass
|
||||
log.warning("LLM server responded with HTTP %s %s", resp.status_code, detail)
|
||||
stream_ctx.__exit__(None, None, None)
|
||||
return None, None
|
||||
|
||||
return None, None # pragma: no cover - the loop always returns
|
||||
|
||||
|
||||
def ai_summary_view() -> flask.Response:
|
||||
"""Stream an AI generated answer for the messages in the request body,
|
||||
response is NDJSON: ``{"delta": ..}`` lines followed by one final
|
||||
@@ -271,22 +324,13 @@ def ai_summary_view() -> flask.Response:
|
||||
# reported as HTTP 502 instead of a line in an already started stream
|
||||
user_api_key = str(sxng_request.preferences.get_value("ai_summary_api_key") or "").strip()
|
||||
client = _get_client(server, cfg, _server_api_key(cfg, server, user_api_key))
|
||||
stream_ctx = client.stream("POST", "/chat/completions", json=chat_payload)
|
||||
upstream = None
|
||||
try:
|
||||
upstream = stream_ctx.__enter__() # pylint: disable=unnecessary-dunder-call
|
||||
if upstream.status_code != 200:
|
||||
stream_ctx.__exit__(None, None, None)
|
||||
upstream = None
|
||||
except httpx.HTTPError:
|
||||
upstream = None
|
||||
if upstream is None:
|
||||
stream_ctx, upstream = _open_upstream(client, chat_payload)
|
||||
if stream_ctx is None or upstream is None:
|
||||
client.close()
|
||||
return flask.Response(json.dumps({"error": "upstream error"}), status=502, mimetype="application/json")
|
||||
|
||||
# from here on nothing must be read from the request context, the
|
||||
# generator runs after the request context has been torn down
|
||||
log = logging.getLogger("searx.plugins.ai_summary")
|
||||
|
||||
def ndjson(obj: dict[str, t.Any]) -> bytes:
|
||||
# the generator bypasses flask's response encoding (direct_passthrough)
|
||||
|
||||
Reference in New Issue
Block a user