diff --git a/docs/tools.md b/docs/tools.md index 96203e60f9..95c5295df3 100644 --- a/docs/tools.md +++ b/docs/tools.md @@ -139,8 +139,14 @@ defaults apply. In hybrid mode, the control plane resolves the policy instead. ## Web search Otari reaches a licensed search API directly. Set `web_search_provider` to -`tavily` or `brave` and `web_search_provider_api_key` to that provider's key; -the key stays in the gateway process and never reaches a caller. +`tavily`, `brave` or `serply` and `web_search_provider_api_key` to that +provider's key; the key stays in the gateway process and never reaches a +caller. + +`tavily` returns extracted page text with its hits, so a search skips the +fetch-and-extract pass the other two need. `brave` and `serply` return snippets, +and `serply` reads Google's index. All three take `time_range` in +`provider_options` as `day`, `week`, `month` or `year`. For evaluation, the bundled SearXNG backend needs no key: diff --git a/src/gateway/core/config.py b/src/gateway/core/config.py index 89a756f3cb..80a93d479c 100644 --- a/src/gateway/core/config.py +++ b/src/gateway/core/config.py @@ -142,7 +142,7 @@ # service. Declared here for the same reason as SEARCH_PROVIDERS above: startup # validation rejects an unknown ``web_search_provider`` without the config layer # importing `gateway.services.web_search_providers`, which imports this name. -WEB_SEARCH_PROVIDERS = ("tavily", "brave") +WEB_SEARCH_PROVIDERS = ("tavily", "brave", "serply") # Providers that authenticate with an API key, so a tool declaring one of them # without a key is a misconfiguration. A SearXNG-shaped backend is normally # keyless (the bundled container, a self-hosted adapter), which is why the key @@ -1126,7 +1126,7 @@ class GatewayConfig(BaseSettings): web_search_provider: str | None = Field( default=None, description=( - "Licensed search API the web-search backend calls directly ('tavily' or 'brave'), " + "Licensed search API the web-search backend calls directly ('tavily', 'brave' or 'serply'), " "instead of the SearXNG-shaped service web_search_url names. Requires " "web_search_provider_api_key. When both are set, web_search_url is not needed." ), diff --git a/src/gateway/services/web_search_providers.py b/src/gateway/services/web_search_providers.py index beaa315489..ece1229c32 100644 --- a/src/gateway/services/web_search_providers.py +++ b/src/gateway/services/web_search_providers.py @@ -12,7 +12,7 @@ * **Standalone.** ``WebSearchBackend`` calls :func:`provider_search` directly when ``web_search_provider`` is configured, so a self-hosted deployment - reaches Tavily or Brave with no adapter container, no second URL, and no + reaches a licensed provider with no adapter container, no second URL, and no extra hop. * **Hosted.** The data plane is a separate process on separate hardware, and a deployment-owned search key must not be on it. There the control plane serves @@ -40,16 +40,26 @@ TAVILY_PROVIDER = "tavily" BRAVE_PROVIDER = "brave" +SERPLY_PROVIDER = "serply" _TAVILY_ENDPOINT = "https://api.tavily.com/search" _BRAVE_ENDPOINT = "https://api.search.brave.com/res/v1/web/search" +_SERPLY_ENDPOINT = "https://api.serply.io/v1/search" -# Both providers document 20 as the ceiling on a page of results, and both treat -# more as an error rather than clamping, so a request is clamped here. ``options`` -# is the opaque ``provider_options`` bag, which can name any number at all. +# Tavily and Brave document 20 as the ceiling on a page of results, and both +# treat more as an error rather than clamping, so a request is clamped here. +# ``options`` is the opaque ``provider_options`` bag, which can name any number +# at all. Serply's page is smaller and is bounded separately, see _SERPLY_PAGE_SIZE. _MAX_RESULTS_CEILING = 20 _BRAVE_DEFAULT_COUNT = 10 +# Serply serves one page of Google results and reads ``num`` as an approximate +# bound rather than a count: ``num=20`` answers with the same page as ``num=10``, +# and a page can come back holding one row more than was asked for. So the +# request is clamped to the page size and the answer is trimmed to it, which a +# provider that returns at most what it was asked for does not need. +_SERPLY_PAGE_SIZE = 10 + # Brave spells recency as ``freshness``. Both the single and the plural forms of # each window are accepted, matching Tavily's own ``time_range`` vocabulary, so # one workspace's ``provider_options`` reads the same whichever provider serves @@ -80,6 +90,19 @@ # caller-supplied default, and :func:`_bounded_max_results` applies both. _TAVILY_OPTION_KEYS = ("max_results", "search_depth", "topic", "time_range", "include_answer") +# Serply passes Google's own ``tbs`` window through, so the same recency +# vocabulary maps onto it rather than onto a spelling of its own. +_SERPLY_TBS = { + "d": "qdr:d", + "day": "qdr:d", + "w": "qdr:w", + "week": "qdr:w", + "m": "qdr:m", + "month": "qdr:m", + "y": "qdr:y", + "year": "qdr:y", +} + class WebSearchProviderError(RuntimeError): """The search provider could not be reached or returned malformed data.""" @@ -111,21 +134,25 @@ async def provider_search( return await _search_tavily(api_key, query, options or {}, client, timeout_s) if normalized == BRAVE_PROVIDER: return await _search_brave(api_key, query, options or {}, client, timeout_s) + if normalized == SERPLY_PROVIDER: + return await _search_serply(api_key, query, options or {}, client, timeout_s) msg = f"web_search_provider must be one of {sorted(WEB_SEARCH_PROVIDERS)}, got '{provider}'" raise ValueError(msg) -def _bounded_max_results(options: Mapping[str, Any], default: int) -> int: +def _bounded_max_results(options: Mapping[str, Any], default: int, ceiling: int = _MAX_RESULTS_CEILING) -> int: """The number of hits to ask the provider for, bounded at what it will serve. ``default`` is the caller's own resolved ceiling, so a deployment that raised ``web_search_max_results`` asks the provider for that many instead of taking the provider's default and being trimmed to fewer by a post-hoc slice. + ``ceiling`` is the page the provider will actually serve, which is not the + same number for every provider. """ requested = options.get("max_results") if isinstance(requested, int) and not isinstance(requested, bool) and requested > 0: - return min(requested, _MAX_RESULTS_CEILING) - return min(max(default, 1), _MAX_RESULTS_CEILING) + return min(requested, ceiling) + return min(max(default, 1), ceiling) async def _search_tavily( @@ -237,6 +264,58 @@ async def _search_brave( return results +async def _search_serply( + api_key: str, + query: str, + options: Mapping[str, Any], + client: httpx.AsyncClient, + timeout_s: float, +) -> list[dict[str, Any]]: + """Serply's Google SERP proxy, which returns snippets only. + + ``extracted_content`` is left unset for the same reason as Brave's, so the + caller fetches and extracts each page itself. Ads and knowledge-graph + entries arrive in sibling keys of the envelope that this does not read, but + a local pack is sometimes folded into ``results`` instead, as a row that + calls itself organic and links back into Google rather than at a site. + """ + limit = _bounded_max_results(options, _SERPLY_PAGE_SIZE, ceiling=_SERPLY_PAGE_SIZE) + params: dict[str, Any] = {"q": query, "num": limit} + tbs = _SERPLY_TBS.get(str(options.get("time_range") or "").strip().lower()) + if tbs is not None: + params["tbs"] = tbs + + payload = await _request( + SERPLY_PROVIDER, + client, + "GET", + _SERPLY_ENDPOINT, + params=params, + headers={"X-Api-Key": api_key, "Accept": "application/json"}, + timeout_s=timeout_s, + ) + + hits = payload.get("results") + if not isinstance(hits, list): + msg = "serply search returned no results list" + raise WebSearchProviderError(msg) + + results: list[dict[str, Any]] = [] + for hit in hits: + if not isinstance(hit, dict) or not hit.get("link"): + continue + results.append( + { + "url": str(hit["link"]), + "title": str(hit.get("title", "")), + "content": str(hit.get("description", "")), + } + ) + if len(results) == limit: + break + return results + + async def _request( provider: str, client: httpx.AsyncClient, diff --git a/tests/unit/test_web_search_providers.py b/tests/unit/test_web_search_providers.py index 3f9e14483a..a56867dd26 100644 --- a/tests/unit/test_web_search_providers.py +++ b/tests/unit/test_web_search_providers.py @@ -1,4 +1,4 @@ -"""Unit tests for the first-party Tavily and Brave web-search providers. +"""Unit tests for the first-party Tavily, Brave and Serply web-search providers. Covers the translation both ways (a query onto each provider's native request, its answer back onto SearXNG-shaped hits) and the failure modes that used to be @@ -18,6 +18,7 @@ TAVILY_HOST = "api.tavily.com" BRAVE_HOST = "api.search.brave.com" +SERPLY_HOST = "api.serply.io" class _Recorder(httpx.AsyncBaseTransport): @@ -61,6 +62,21 @@ def _client(response: httpx.Response | Exception) -> tuple[httpx.AsyncClient, _R } } +SERPLY_OK = { + "results": [ + {"link": "https://example.com/a", "title": "A", "description": "snippet a", "result_type": "organic"}, + {"title": "no link", "description": "dropped"}, + ] +} + +# Serply answers ``num=10`` with whatever the SERP page holds, which is sometimes +# one row more. Measured live: "buy running shoes" returns 11. +SERPLY_OVERLONG = { + "results": [ + {"link": f"https://example.com/{index}", "title": str(index), "description": "s"} for index in range(11) + ] +} + BRAVE_DATED = { "web": { "results": [ @@ -264,3 +280,64 @@ async def test_a_caller_ceiling_reaches_the_provider(provider: str) -> None: request = recorder.requests[0] asked = json.loads(request.content)["max_results"] if provider == "tavily" else request.url.params["count"] assert int(asked) == 15 + + +@pytest.mark.asyncio +async def test_serply_maps_results_and_leaves_extraction_to_the_caller() -> None: + client, recorder = _client(httpx.Response(200, json=SERPLY_OK)) + async with client: + results = await provider_search(provider="serply", api_key="srp-x", query="claude code", client=client) + + assert results == [{"url": "https://example.com/a", "title": "A", "content": "snippet a"}] + request = recorder.requests[0] + assert request.url.host == SERPLY_HOST + assert request.headers["x-api-key"] == "srp-x" + + +@pytest.mark.asyncio +async def test_serply_trims_a_page_that_came_back_longer_than_asked() -> None: + """``num`` bounds the page Serply builds, it does not cap what the page holds. + + Without the trim a deployment's ``web_search_max_results`` would be a number + the model's result block is allowed to exceed. + """ + client, _ = _client(httpx.Response(200, json=SERPLY_OVERLONG)) + async with client: + results = await provider_search( + provider="serply", api_key="srp-x", query="q", options={"max_results": 3}, client=client + ) + + assert len(results) == 3 + + +@pytest.mark.asyncio +async def test_serply_clamps_max_results_to_the_page_it_will_serve() -> None: + """Serply serves one SERP page, and asking for more than it holds returns the + same page rather than an error, so the bound is ours to apply.""" + client, recorder = _client(httpx.Response(200, json=SERPLY_OK)) + async with client: + await provider_search( + provider="serply", api_key="srp-x", query="q", options={"max_results": 500}, client=client + ) + + assert recorder.requests[0].url.params["num"] == "10" + + +@pytest.mark.parametrize(("time_range", "expected"), [("day", "qdr:d"), ("w", "qdr:w"), ("Month", "qdr:m")]) +@pytest.mark.asyncio +async def test_serply_sends_time_range_as_a_google_window(time_range: str, expected: str) -> None: + client, recorder = _client(httpx.Response(200, json=SERPLY_OK)) + async with client: + await provider_search( + provider="serply", api_key="srp-x", query="q", options={"time_range": time_range}, client=client + ) + + assert recorder.requests[0].url.params["tbs"] == expected + + +@pytest.mark.asyncio +async def test_serply_missing_results_list_is_an_error_not_an_empty_search() -> None: + client, _ = _client(httpx.Response(200, json={"query": "q", "total": 0})) + async with client: + with pytest.raises(WebSearchProviderError): + await provider_search(provider="serply", api_key="srp-x", query="q", client=client)