From 420b4a2797fa283cd15b5ee267380743115eca5f Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Sat, 26 Sep 2026 05:28:13 +0200 Subject: [PATCH] fix(retrieval): name the link when process/web cannot fetch it (#31351) A link whose host refuses the connection, such as a closed port, still came back from POST /api/v1/retrieval/process/web as "Error querying knowledge base", so the caller was told the knowledge base failed when the link was the problem. The web loaders log a failed fetch and return no documents. That empty result then failed while being saved to the vector store, and the save error was the one reported. process_web now answers with the existing "Could not read content from " 400 as soon as the loader returns no documents, the same message a link refused by the fetch filter already gets. The check sits in the endpoint so web search and the other users of the loaders keep their current behaviour. With process=false or embedding bypassed, an unreachable link now gets the same 400 where it used to return 200 with empty content. Fixes #31347 --- backend/open_webui/routers/retrieval.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/backend/open_webui/routers/retrieval.py b/backend/open_webui/routers/retrieval.py index b9375e56c6..6517ed0674 100644 --- a/backend/open_webui/routers/retrieval.py +++ b/backend/open_webui/routers/retrieval.py @@ -2457,6 +2457,13 @@ async def process_web( detail=ERROR_MESSAGES.DEFAULT(e, f'Could not read content from {form_data.url}'), ) + # web loaders swallow fetch errors and return no documents + if not docs: + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, + detail=ERROR_MESSAGES.DEFAULT(f'Could not read content from {form_data.url}'), + ) + try: log.debug('text_content: %s', content)