From 681ec1d9ad8ca228518cc81df6ca95ad44f376ee Mon Sep 17 00:00:00 2001 From: WYXNICK <88577248+WYXNICK@users.noreply.github.com> Date: Sun, 20 Sep 2026 09:37:21 +0800 Subject: [PATCH 1/2] feat(deepeyes): add pluggable search backends MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit # ⭐ Feature - Add deterministic mock, Search-R1 and configurable external search. - Normalize results and preserve non-terminal errors with bounded retries. - Forward search settings to Ray workers without tracing credentials. - Document backend configuration and deployment requirements. # ✅ Tests - Cover HTTP adapters, failure handling, agent continuation and launchers. - Pass all 86 DeepEyes-V2 tests and all repository pre-commit checks. - Live search services and GPU training were not available for validation. --- examples/deepeyes_v2_agentic/README.md | 100 +++++ .../app/env_deepeyes_v2.py | 7 +- .../app/search_backends.py | 239 +++++++++++ .../deepeyes_v2_agentic/app/search_utils.py | 50 +-- examples/deepeyes_v2_agentic/env.sh.example | 9 + .../run_deepeyes_v2_agentic.sh | 21 + .../run_deepeyes_v2_agentic_klx.sh | 20 + .../test_env_extractors.py | 10 + .../test_search_backends.py | 394 ++++++++++++++++++ .../test_search_launchers.py | 83 ++++ 10 files changed, 895 insertions(+), 38 deletions(-) create mode 100644 examples/deepeyes_v2_agentic/app/search_backends.py create mode 100644 tests/examples/deepeyes_v2_agentic/test_search_backends.py create mode 100644 tests/examples/deepeyes_v2_agentic/test_search_launchers.py diff --git a/examples/deepeyes_v2_agentic/README.md b/examples/deepeyes_v2_agentic/README.md index 04541466c..7966d2a7a 100644 --- a/examples/deepeyes_v2_agentic/README.md +++ b/examples/deepeyes_v2_agentic/README.md @@ -105,6 +105,106 @@ and propagates it to every Ray worker via `--runtime-env-json`. Set `APPTAINER_IMAGE_PATH` explicitly to override (e.g. shared NFS path for multi-node). +## Text search + +Text search defaults to a deterministic, offline mock. No search service, network +or API key is needed; model inference and the Python sandbox retain their normal +requirements. Select a real backend with one YAML file: + +```bash +export DEEPEYES_V2_SEARCH_CONFIG=/path/to/search.yaml +``` + +For a Search-R1-compatible service: + +```yaml +backend: retriever +retriever: + url: http://your-retriever:8000/retrieve +``` + +The client sends `POST {"queries": [query], "topk": k, "return_scores": true}` +and accepts both raw corpus documents and `document`/`score` wrappers inside +`result[0]`. It preserves `contents` as the snippet. Missing titles use the first +line of the passage; missing links remain empty rather than inventing URLs. + +For [Tavily](https://docs.tavily.com/documentation/api-reference/endpoint/search), +set `DEEPEYES_V2_SEARCH_API_KEY` in your environment and use: + +```yaml +backend: external +external: + endpoint: https://api.tavily.com/search + method: POST + auth: + header: Authorization + prefix: "Bearer " + request_map: + query: query + size: max_results + response_map: + results: results + title: title + link: url + snippet: content + date: published_date +``` + +`external` supports GET query parameters or POST JSON, configurable header +authentication, and dot-separated response paths. Omit `auth` for an unauthenticated +endpoint and omit the date mapping when the API does not provide one. Request +mapping names must be distinct; no provider SDK is required. + +Common settings (all optional): + +| Setting | Default | Meaning | +| ------------- | ------- | --------------------------------------------------------------- | +| `backend` | `mock` | `mock`, `retriever`, or `external` | +| `top_k` | `5` | Maximum results; an explicit `search(query, size)` overrides it | +| `timeout_s` | `10` | HTTP connect/read timeout, not a total wall-clock deadline | +| `max_retries` | `2` | Retries after the first request; zero means one attempt | +| `backoff_s` | `0.5` | Initial retry delay; doubles up to 2 seconds | +| `trust_env` | `false` | Enable environment proxy/netrc settings when needed | + +Successful calls return `elapsed_time` in seconds and `data`, a list of +`{title, link, snippet, date}` records. Date is a string or null. Empty results +are successful; malformed responses are errors. Connection failures, timeouts, +HTTP 408/429 and 5xx statuses are retried. Other HTTP failures (including +redirects), malformed JSON and configuration errors fail immediately. Failure +returns the existing `"Error"` sentinel, which becomes a non-terminal +`search_failed` observation. A broken real backend never falls back to mock. +Diagnostic logs identify the error without printing credentials or response bodies. + +Both training launchers forward the configuration path and API key to Ray workers. +The YAML must be readable at that path on every node; forwarding a path does not +upload its file. Credentials are excluded from shell tracing, but Ray runtime +configuration and submission process arguments may be visible to cluster +administrators. Use worker-provisioned credentials if that exposure is unsuitable. + +CPU tests use a local HTTP server for the Search-R1 and external protocols, plus a +scripted model to exercise the real Agent loop. They need no search service, GPU +or Apptainer instance: + +```bash +python -m pytest tests/examples/deepeyes_v2_agentic -q +``` + +To check your configured service separately, from the repository root: + +```bash +PYTHONPATH=examples/deepeyes_v2_agentic:. python - <<'PYTHON' +from app.search_utils import search + +result = search("What is reinforcement learning?", size=3) +assert result != "Error", "See the search diagnostic above" +print(result) +PYTHON +``` + +The protocol tests do not validate a live E5/FAISS deployment or API credentials. +The repository's native Search-R1 server uses CUDA; the search client itself can +run on a CPU host, including macOS. + ## Image-search cache (optional, only for the `search` split) The `image_search` branch hits a precomputed diff --git a/examples/deepeyes_v2_agentic/app/env_deepeyes_v2.py b/examples/deepeyes_v2_agentic/app/env_deepeyes_v2.py index 443b77ab2..2d21dd4f8 100644 --- a/examples/deepeyes_v2_agentic/app/env_deepeyes_v2.py +++ b/examples/deepeyes_v2_agentic/app/env_deepeyes_v2.py @@ -444,7 +444,7 @@ def _dispatch_search(self, tool_name: str, tool_args: Any) -> dict: } # tool_name == "search" - query = tool_args["query"] if isinstance(tool_args, dict) and "query" in tool_args else str(tool_args) + query = tool_args.get("query") if isinstance(tool_args, dict) else tool_args result = search(query) if result == "Error": return {"status": "error", "result": "Error", "images": []} @@ -458,9 +458,8 @@ def _dispatch_search(self, tool_name: str, tool_args: Any) -> dict: if page.get("snippet") is not None: snippet = "\n" + page["snippet"] snippets.append(f"{idx + 1}. [{page['title']}]({page['link']}){date_published}{snippet}") - content = ( - f"A Google search for '{query}' found {len(snippets)} results:" - f"\n\n## Web Results\n" + "\n\n".join(snippets) + content = f"A Web search for '{query}' found {len(snippets)} results:\n\n## Web Results\n" + "\n\n".join( + snippets ) except (KeyError, TypeError) as exc: return { diff --git a/examples/deepeyes_v2_agentic/app/search_backends.py b/examples/deepeyes_v2_agentic/app/search_backends.py new file mode 100644 index 000000000..a96e1eb3a --- /dev/null +++ b/examples/deepeyes_v2_agentic/app/search_backends.py @@ -0,0 +1,239 @@ +# Copyright (c) 2026 Relax Authors. All Rights Reserved. + +"""Text search adapters configured by DEEPEYES_V2_SEARCH_CONFIG.""" + +from __future__ import annotations + +import math +import os +import time +from pathlib import Path +from typing import Any +from urllib.parse import urlsplit + + +DEFAULTS = { + "backend": "mock", + "top_k": 5, + "timeout_s": 10.0, + "max_retries": 2, + "backoff_s": 0.5, + "trust_env": False, +} + + +class SearchError(ValueError): + """Search failure whose message is safe to log.""" + + +def _mapping(value: Any, name: str, allowed: set[str]) -> dict: + if not isinstance(value, dict): + raise SearchError(f"{name} must be a mapping") + if value.keys() - allowed: + raise SearchError(f"{name} contains unknown fields") + return value + + +def _text(value: Any, name: str, *, empty: bool = False) -> str: + if not isinstance(value, str) or (not empty and not value.strip()): + raise SearchError(f"{name} must be a {'non-empty ' if not empty else ''}string") + return value + + +def _integer(value: Any, name: str, minimum: int) -> int: + if type(value) is not int or value < minimum: + raise SearchError(f"{name} must be an integer >= {minimum}") + return value + + +def _load_config() -> dict: + config = dict(DEFAULTS) + path = os.environ.get("DEEPEYES_V2_SEARCH_CONFIG", "").strip() + if path: + import yaml + + try: + source = yaml.safe_load(Path(path).expanduser().read_text(encoding="utf-8")) + except (OSError, UnicodeError, yaml.YAMLError) as exc: + raise SearchError(f"cannot read DEEPEYES_V2_SEARCH_CONFIG ({type(exc).__name__})") from None + config.update(_mapping(source, "search config", set(DEFAULTS) | {"retriever", "external"})) + if config["backend"] not in ("mock", "retriever", "external"): + raise SearchError("backend must be mock, retriever or external") + _integer(config["top_k"], "top_k", 1) + _integer(config["max_retries"], "max_retries", 0) + for name in ("timeout_s", "backoff_s"): + value = config[name] + if type(value) not in (int, float) or not math.isfinite(value) or value < 0: + raise SearchError(f"{name} must be a finite non-negative number") + if config["timeout_s"] == 0: + raise SearchError("timeout_s must be positive") + if not isinstance(config["trust_env"], bool): + raise SearchError("trust_env must be boolean") + return config + + +def _endpoint(value: Any, name: str) -> str: + value = _text(value, name).strip() + try: + parsed = urlsplit(value) + valid = parsed.scheme in ("http", "https") and parsed.hostname and not parsed.username and not parsed.password + parsed.port + except ValueError: + valid = False + if not valid: + raise SearchError(f"{name} must be an HTTP(S) URL without credentials") + return value + + +def _request(config: dict, url: str, method: str, payload: dict, headers: dict) -> Any: + import requests + + delay = min(config["backoff_s"], 2.0) + attempts = config["max_retries"] + 1 + with requests.Session() as session: + session.trust_env = config["trust_env"] + for attempt in range(attempts): + try: + with session.request( + method, + url, + headers=headers, + **{"params" if method == "GET" else "json": payload}, + timeout=config["timeout_s"], + allow_redirects=False, + ) as response: + status = response.status_code + if 200 <= status < 300: + try: + return response.json() + except ValueError: + raise SearchError("search response is not valid JSON") from None + if status not in (408, 429) and not 500 <= status < 600: + raise SearchError(f"search HTTP status {status}") + reason = f"HTTP {status}" + except (requests.Timeout, requests.ConnectionError) as exc: + reason = type(exc).__name__ + except requests.RequestException as exc: + raise SearchError(f"search request failed ({type(exc).__name__})") from None + if attempt + 1 < attempts: + time.sleep(delay) + delay = min(delay * 2, 2.0) + raise SearchError(f"search failed after {attempts} attempts ({reason})") + + +def _date(value: Any) -> str | None: + return None if value is None else _text(value, "result.date", empty=True) + + +def _retriever_rows(body: Any, size: int) -> list[dict]: + batches = body.get("result") if isinstance(body, dict) else None + if not isinstance(batches, list) or len(batches) != 1 or not isinstance(batches[0], list): + raise SearchError("retriever response must contain one result batch") + rows = [] + for item in batches[0][:size]: + doc = item.get("document", item) if isinstance(item, dict) else None + if not isinstance(doc, dict): + raise SearchError("retriever document must be a mapping") + content = _text(doc.get("contents"), "retriever document.contents") + title = doc.get("title") + if title is not None: + title = _text(title, "result.title", empty=True) + title = title or content.strip().splitlines()[0].strip().strip('"') + link = doc.get("url") + if link is None: + link = doc.get("link", "") + rows.append( + { + "title": _text(title, "result.title", empty=True), + "link": _text("" if link is None else link, "result.link", empty=True), + "snippet": content, + "date": _date(doc.get("date")), + } + ) + return rows + + +def _field(row: Any, path: str) -> Any: + for key in path.split("."): + if not isinstance(row, dict): + return None + row = row.get(key) + return row + + +def _external(config: dict, query: str, size: int) -> list[dict]: + external = _mapping( + config.get("external"), "external", {"endpoint", "method", "auth", "request_map", "response_map"} + ) + url = _endpoint(external.get("endpoint"), "external.endpoint") + method = _text(external.get("method", "POST"), "external.method").upper() + if method not in ("GET", "POST"): + raise SearchError("external.method must be GET or POST") + request_map = _mapping(external.get("request_map"), "external.request_map", {"query", "size"}) + response_map = _mapping( + external.get("response_map"), "external.response_map", {"results", "title", "link", "snippet", "date"} + ) + for name in ("query", "size"): + _text(request_map.get(name), f"external.request_map.{name}") + if request_map["query"] == request_map["size"]: + raise SearchError("external.request_map fields must be distinct") + for name in ("results", "title", "link", "snippet"): + _text(response_map.get(name), f"external.response_map.{name}") + if "date" in response_map: + _text(response_map["date"], "external.response_map.date") + headers = {"Accept": "application/json"} + if "auth" in external: + auth = _mapping(external["auth"], "external.auth", {"header", "prefix"}) + header = _text(auth.get("header"), "external.auth.header") + prefix = _text(auth.get("prefix", ""), "external.auth.prefix", empty=True) + secret = os.environ.get("DEEPEYES_V2_SEARCH_API_KEY", "") + if not secret.strip() or any(c in header + prefix + secret for c in "\r\n"): + raise SearchError("external authentication is missing or invalid") + headers[header] = prefix + secret + body = _request(config, url, method, {request_map["query"]: query, request_map["size"]: size}, headers) + items = _field(body, response_map["results"]) + if not isinstance(items, list): + raise SearchError("external results must be a list") + rows = [] + for item in items[:size]: + row = { + name: _text(_field(item, response_map[name]), f"result.{name}", empty=True) + for name in ("title", "link", "snippet") + } + row["date"] = _date(_field(item, response_map["date"])) if "date" in response_map else None + rows.append(row) + return rows + + +def run_search(query: str, size: int | None = None) -> dict: + """Return normalized search results, or raise SearchError on failure.""" + query = _text(query, "query").strip() + config = _load_config() + size = _integer(config["top_k"] if size is None else size, "size", 1) + if config["backend"] == "mock": + return { + "elapsed_time": 0.0, + "data": [ + { + "title": f"Offline mock result {i + 1}", + "link": f"https://example.invalid/{i + 1}", + "snippet": f"Deterministic offline result for: {query}", + "date": None, + } + for i in range(size) + ], + } + started = time.monotonic() + if config["backend"] == "retriever": + retriever = _mapping(config.get("retriever"), "retriever", {"url"}) + body = _request( + config, + _endpoint(retriever.get("url"), "retriever.url"), + "POST", + {"queries": [query], "topk": size, "return_scores": True}, + {"Accept": "application/json"}, + ) + rows = _retriever_rows(body, size) + else: + rows = _external(config, query, size) + return {"elapsed_time": time.monotonic() - started, "data": rows} diff --git a/examples/deepeyes_v2_agentic/app/search_utils.py b/examples/deepeyes_v2_agentic/app/search_utils.py index 1f501a95a..a37e16fbc 100644 --- a/examples/deepeyes_v2_agentic/app/search_utils.py +++ b/examples/deepeyes_v2_agentic/app/search_utils.py @@ -2,8 +2,7 @@ """Search tool helpers for the DeepEyesV2 env. -* :func:`search` is a placeholder web-search returning canned snippets so the - recipe runs end-to-end without a real backend. +* :func:`search` supports offline mock, Search-R1 and external HTTP backends. * :func:`image_search` serves cached results keyed by ``data_idx`` from JSON files listed in ``DEEPEYES_V2_SEARCH_CACHE_PATHS`` (colon/comma-separated). Missing / unparsable caches degrade to returning ``"Error"`` so the env @@ -13,13 +12,15 @@ from __future__ import annotations import json -import logging import os -import random -import time +from typing import Literal +from relax.utils.logging_utils import get_logger -logger = logging.getLogger(__name__) +from .search_backends import SearchError, run_search + + +logger = get_logger(__name__) def _load_image_search_cache() -> dict: @@ -63,34 +64,15 @@ def _get_image_search_cache() -> dict: return _IMAGE_SEARCH_CACHE -def search(query: str, size: int = 5): - """Web-search placeholder. Returns canned snippets in the shape:: - - {"elapsed_time": float, "data": [{"title", "link", "snippet", "date"?}, ...]} - - Replace with a real backend (Serper / Google / Bing / internal) for - production training. - """ - max_try = 3 - result = "Error" - for try_idx in range(max_try): - try: - result = {"elapsed_time": 0.0, "data": []} - for i in range(size): - result["data"].append( - { - "snippet": f"This is a placeholder snippet for query: {query}", - "title": f"Placeholder Title {i}", - "link": f"http://example.com/{i}", - } - ) - break - except Exception as e: - logger.warning(f"[search] attempt {try_idx + 1}/{max_try} failed: {e}") - result = "Error" - if try_idx < max_try - 1: - time.sleep((try_idx + 1) * random.randint(1, 5)) - return result +def search(query: str, size: int | None = None) -> dict | Literal["Error"]: + """Search with the configured backend; failures remain non-terminal.""" + try: + return run_search(query, size) + except SearchError as exc: + logger.warning(f"[search] {exc}") + except Exception as exc: + logger.warning(f"[search] unexpected failure ({type(exc).__name__})") + return "Error" def image_search(_query, data_idx: str | None = None): diff --git a/examples/deepeyes_v2_agentic/env.sh.example b/examples/deepeyes_v2_agentic/env.sh.example index 694c0902a..c0c9e3088 100644 --- a/examples/deepeyes_v2_agentic/env.sh.example +++ b/examples/deepeyes_v2_agentic/env.sh.example @@ -57,3 +57,12 @@ export DEEPEYES_V2_APP_ENV_ROOT=/tmp/deepeyes-v2-app-env # verified the mirror serves the full dataset (some mirrors 308-redirect # large files back to upstream, which huggingface_hub refuses). # export HF_ENDPOINT=https://your.hf.mirror + +# ============================================================================ +# Optional — text search (default: deterministic offline mock) +# ============================================================================ +# One YAML file selects mock/retriever/external; see README for examples. +# The file must be readable at this path on every worker node. +# export DEEPEYES_V2_SEARCH_CONFIG=/path/to/search.yaml +# For an external backend with auth, set DEEPEYES_V2_SEARCH_API_KEY in the +# environment or this gitignored file. Do not commit credentials. diff --git a/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic.sh b/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic.sh index cd822296c..5a79ef599 100755 --- a/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic.sh +++ b/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic.sh @@ -13,7 +13,9 @@ TIMESTAMP=$(date "+%Y-%m-%d-%H:%M:%S") SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" # Auto-source env.sh if present (gitignored, machine-specific overrides). # shellcheck source=/dev/null +set +x [ -f "${SCRIPT_DIR}/env.sh" ] && source "${SCRIPT_DIR}/env.sh" +set -x if [ -z "${RELAX_ENTRYPOINT_MODE:-}" ]; then source "${SCRIPT_DIR}/../../scripts/entrypoint/local.sh" @@ -108,6 +110,8 @@ NUM_ROLLOUT="${NUM_ROLLOUT:=2000}" # agent process can find apptainer / search cache. # SANDBOX_CONFIG_PATH is required — the agent reads it in _build_executor # to find the apptainer backend YAML config (image path, bind paths, etc). +# Runtime configuration may include credentials. +set +x RUNTIME_ENV_JSON=$(cat <&1 | tee logs/${EXP_NAME}.log +set -x diff --git a/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic_klx.sh b/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic_klx.sh index 9c40c4a6b..e74642ef1 100644 --- a/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic_klx.sh +++ b/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic_klx.sh @@ -44,7 +44,9 @@ TIMESTAMP=$(date "+%Y-%m-%d-%H:%M:%S") SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" # Auto-source env.sh if present (gitignored, machine-specific overrides). # shellcheck source=/dev/null +set +x [ -f "${SCRIPT_DIR}/env.sh" ] && source "${SCRIPT_DIR}/env.sh" +set -x if [ -z "${RELAX_ENTRYPOINT_MODE:-}" ]; then source "${SCRIPT_DIR}/../../scripts/entrypoint/local.sh" @@ -139,6 +141,8 @@ NUM_ROLLOUT="${NUM_ROLLOUT:=2000}" # agent process can find apptainer / search cache. # SANDBOX_CONFIG_PATH is required — the agent reads it in _build_executor # to find the apptainer backend YAML config (image path, bind paths, etc). +# Runtime configuration may include credentials. +set +x EXTRA_ENV_VARS_JSON="\"SANDBOX_BACKEND\": \"apptainer_jupyter\", \"SANDBOX_CONFIG_PATH\": \"${SCRIPT_DIR}/apptainer_env/apptainer_config.yaml\", \"APPTAINER_IMAGE_PATH\": \"${APPTAINER_IMAGE_PATH}\", @@ -150,6 +154,7 @@ EXTRA_ENV_VARS_JSON="\"SANDBOX_BACKEND\": \"apptainer_jupyter\", \"XMLIR_ENABLE_H2D_SSE_COPY\": \"${XMLIR_ENABLE_H2D_SSE_COPY:-1}\", \"USE_CAST_FC_FUSION\": \"${USE_CAST_FC_FUSION:-1}\"" source "${SCRIPT_DIR}/../../scripts/entrypoint/runtime-env-klx.sh" +set -x ROLLOUT_ARGS=( --prompt-data "${PROMPT_SET}" @@ -331,6 +336,21 @@ print(json.dumps(payload)) PY ) fi +# Pass search configuration to Ray workers without tracing credentials. +set +x +export RUNTIME_ENV_JSON +RUNTIME_ENV_JSON=$(python3 - <<'PYTHON' +import json +import os + +payload = json.loads(os.environ["RUNTIME_ENV_JSON"]) +for name in ("DEEPEYES_V2_SEARCH_CONFIG", "DEEPEYES_V2_SEARCH_API_KEY"): + value = os.environ.get(name) + if value: + payload.setdefault("env_vars", {})[name] = value +print(json.dumps(payload)) +PYTHON +) ray job submit ${RAY_NO_WAIT:+--no-wait} --address="http://127.0.0.1:8265" \ --runtime-env-json "${RUNTIME_ENV_JSON}" \ -- python3 relax/entrypoints/train.py \ diff --git a/tests/examples/deepeyes_v2_agentic/test_env_extractors.py b/tests/examples/deepeyes_v2_agentic/test_env_extractors.py index 7188def41..abe343e12 100644 --- a/tests/examples/deepeyes_v2_agentic/test_env_extractors.py +++ b/tests/examples/deepeyes_v2_agentic/test_env_extractors.py @@ -7,6 +7,7 @@ from __future__ import annotations +import json import sys from pathlib import Path @@ -55,3 +56,12 @@ def test_encode_image_data_uri_returns_png_data_url(): uri = encode_image_data_uri(Image.new("RGB", (4, 4))) assert uri.startswith("data:image/png;base64,") + + +def test_extract_tool_call_preserves_open_tag_in_json(): + text = '{"name":"search","arguments":{"query":"explain tags"}}' + assert extract_tool_call(text)["arguments"]["query"] == "explain tags" + assert extract_tool_call(text + "{") == extract_tool_call(text) + code = "print('')" + text = "" + json.dumps({"name": "python_exec", "arguments": {"code": code}}) + "" + assert extract_code(text) == code diff --git a/tests/examples/deepeyes_v2_agentic/test_search_backends.py b/tests/examples/deepeyes_v2_agentic/test_search_backends.py new file mode 100644 index 000000000..0f21a93c0 --- /dev/null +++ b/tests/examples/deepeyes_v2_agentic/test_search_backends.py @@ -0,0 +1,394 @@ +# Copyright (c) 2026 Relax Authors. All Rights Reserved. + +"""Search contract and local HTTP integration tests; no external services.""" + +from __future__ import annotations + +import json +import socket +import sys +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from types import SimpleNamespace +from urllib.parse import parse_qs, urlsplit + +import pytest +import requests +import yaml + + +EXAMPLE_DIR = Path(__file__).resolve().parents[3] / "examples" / "deepeyes_v2_agentic" +sys.path.insert(0, str(EXAMPLE_DIR)) + +from app import agent, search_backends, search_utils # noqa: E402 +from app.env_deepeyes_v2 import DeepEyesV2Env # noqa: E402 + + +@pytest.fixture(autouse=True) +def clean_search_env(monkeypatch): + monkeypatch.delenv("DEEPEYES_V2_SEARCH_CONFIG", raising=False) + monkeypatch.delenv("DEEPEYES_V2_SEARCH_API_KEY", raising=False) + + +@pytest.fixture +def configure(tmp_path, monkeypatch): + def write(config): + path = tmp_path / "search.yaml" + path.write_text(yaml.safe_dump(config), encoding="utf-8") + monkeypatch.setenv("DEEPEYES_V2_SEARCH_CONFIG", str(path)) + return path + + return write + + +@pytest.fixture +def server(): + state = SimpleNamespace(requests=[], status=200, body={"result": [[]]}, delay=0.0) + + class Handler(BaseHTTPRequestHandler): + def do_GET(self): + self.respond(parse_qs(urlsplit(self.path).query)) + + def do_POST(self): + self.respond(json.loads(self.rfile.read(int(self.headers["Content-Length"])))) + + def respond(self, payload): + state.requests.append((self.command, payload, dict(self.headers))) + time.sleep(state.delay) + status = state.status.pop(0) if isinstance(state.status, list) else state.status + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Location", "/redirect-target") + self.end_headers() + body = state.body if isinstance(state.body, bytes) else json.dumps(state.body).encode() + try: + self.wfile.write(body) + except (BrokenPipeError, ConnectionResetError): + pass + + def log_message(self, *_args): + pass + + httpd = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + state.url = f"http://127.0.0.1:{httpd.server_port}/retrieve" + thread = threading.Thread(target=httpd.serve_forever, kwargs={"poll_interval": 0.01}) + thread.start() + try: + yield state + finally: + httpd.shutdown() + thread.join() + httpd.server_close() + + +def retriever_config(server, **options): + return {"backend": "retriever", "retriever": {"url": server.url}, "backoff_s": 0, **options} + + +def external_config(server, **options): + return { + "backend": "external", + "backoff_s": 0, + "external": { + "endpoint": server.url, + "method": "POST", + "auth": {"header": "Authorization", "prefix": "Bearer "}, + "request_map": {"query": "query", "size": "max_results"}, + "response_map": { + "results": "results", + "title": "title", + "link": "url", + "snippet": "content", + "date": "published_date", + }, + **options, + }, + } + + +def test_mock_is_deterministic_without_network(monkeypatch): + def forbidden(*args, **kwargs): + pytest.fail("mock attempted network access") + + monkeypatch.setattr(socket, "socket", forbidden) + first = search_utils.search("搜索", 2) + assert first == search_utils.search("搜索", 2) + assert first["elapsed_time"] == 0.0 + assert len(first["data"]) == 2 + assert all(set(row) == {"title", "link", "snippet", "date"} and row["date"] is None for row in first["data"]) + assert "搜索" in first["data"][0]["snippet"] + + +def test_config_top_k_and_explicit_size(configure): + assert len(search_utils.search("q")["data"]) == 5 + configure({"top_k": 2}) + assert len(search_utils.search("q")["data"]) == 2 + assert len(search_utils.search("q", 1)["data"]) == 1 + configure({"top_k": 3}) + assert len(search_utils.search("q")["data"]) == 3 + + +@pytest.mark.parametrize("query", [None, {}, [], "", " ", 42]) +def test_invalid_query_returns_error(query): + assert search_utils.search(query) == "Error" + + +@pytest.mark.parametrize("size", [0, -1, True, 1.5, "2"]) +def test_invalid_size_returns_error(size): + assert search_utils.search("q", size) == "Error" + + +@pytest.mark.parametrize( + "config", + [ + [], + None, + {"backend": "typo"}, + {"backend": []}, + {"top_k": True}, + {"top_k": 0}, + {"timeout_s": 0}, + {"timeout_s": float("nan")}, + {"timeout_s": float("inf")}, + {"max_retries": -1}, + {"max_retries": False}, + {"backoff_s": -1}, + {"trust_env": "false"}, + {"topk": 1}, + {"backend": "retriever", "retriever": {"url": None}}, + {"backend": "retriever", "retriever": {"url": "file:///tmp/data"}}, + {"backend": "retriever", "retriever": {"url": "http://user:secret@host"}}, + ], +) +def test_bad_config_does_not_request(configure, monkeypatch, config): + configure(config) + monkeypatch.setattr( + requests.Session, "request", lambda *a, **kw: pytest.fail("invalid configuration made a request") + ) + assert search_utils.search("q") == "Error" + + +def test_config_read_errors_are_diagnostic_and_redacted(configure, monkeypatch, caplog, tmp_path): + monkeypatch.setenv("DEEPEYES_V2_SEARCH_CONFIG", str(tmp_path / "missing.yaml")) + assert search_utils.search("q") == "Error" + assert "FileNotFoundError" in caplog.text + path = configure({}) + path.write_text("external: [sensitive-value", encoding="utf-8") + assert search_utils.search("q") == "Error" + assert "DEEPEYES_V2_SEARCH_CONFIG" in caplog.text + assert "sensitive-value" not in caplog.text + + +@pytest.mark.parametrize("scored", [False, True]) +def test_search_r1_protocol_and_normalization(server, configure, scored): + docs = [ + {"contents": '"Relax"\nA real passage.', "id": "123"}, + {"title": "Second", "url": "https://example.org", "contents": "More", "date": "2026-09-19"}, + ] + server.body = {"result": [[{"document": doc, "score": 0.9} for doc in docs] if scored else docs]} + configure(retriever_config(server, top_k=2)) + result = search_utils.search("a query") + assert result["data"] == [ + {"title": "Relax", "link": "", "snippet": '"Relax"\nA real passage.', "date": None}, + {"title": "Second", "link": "https://example.org", "snippet": "More", "date": "2026-09-19"}, + ] + assert result["elapsed_time"] >= 0 + assert server.requests[0][:2] == ("POST", {"queries": ["a query"], "topk": 2, "return_scores": True}) + assert len(search_utils.search("q", 1)["data"]) == 1 + + +@pytest.mark.parametrize( + "body", + [ + {}, + {"result": []}, + {"result": [[], []]}, + {"result": [None]}, + {"result": [[None]]}, + {"result": [[{"document": []}]]}, + {"result": [[{"contents": 5}]]}, + {"result": [[{"contents": "text", "date": 12}]]}, + b"not json", + ], +) +def test_malformed_retriever_response_is_not_empty_success(server, configure, body): + server.body = body + configure(retriever_config(server)) + assert search_utils.search("q") == "Error" + assert len(server.requests) == 1 + + +def test_empty_results_are_successful(server, configure): + configure(retriever_config(server)) + assert search_utils.search("q")["data"] == [] + assert len(server.requests) == 1 + + +@pytest.mark.parametrize("method", ["POST", "GET"]) +def test_external_mapping_and_auth(server, configure, monkeypatch, method): + configure(external_config(server, method=method)) + monkeypatch.setenv("DEEPEYES_V2_SEARCH_API_KEY", "test-credential") + server.body = {"results": [{"title": "Page", "url": "https://example.org", "content": "Retrieved text"}]} + result = search_utils.search("中文 & query", 1) + assert result["data"] == [ + {"title": "Page", "link": "https://example.org", "snippet": "Retrieved text", "date": None} + ] + expected = ( + {"query": "中文 & query", "max_results": 1} + if method == "POST" + else {"query": ["中文 & query"], "max_results": ["1"]} + ) + assert server.requests[0][:2] == (method, expected) + assert server.requests[0][2]["Authorization"] == "Bearer test-credential" + + +def test_external_nested_mapping_without_auth(server, configure): + config = external_config(server) + del config["external"]["auth"] + config["external"]["response_map"] = { + "results": "data.items", + "title": "page.title", + "link": "url", + "snippet": "text", + } + configure(config) + server.body = { + "data": {"items": [{"page": {"title": "Nested"}, "url": "https://example.org", "text": "Actual text"}]} + } + assert search_utils.search("q")["data"][0]["title"] == "Nested" + assert "Authorization" not in server.requests[0][2] + + +@pytest.mark.parametrize( + "field,value", + [ + ("method", 5), + ("endpoint", None), + ("request_map", {"query": "q", "size": "q"}), + ("response_map", {"results": "results"}), + ("auth", {"header": "Authorization", "prefix": "\n"}), + ], +) +def test_invalid_external_config_is_rejected_before_http(server, configure, monkeypatch, field, value): + monkeypatch.setenv("DEEPEYES_V2_SEARCH_API_KEY", "test-credential") + configure(external_config(server, **{field: value})) + assert search_utils.search("q") == "Error" + assert not server.requests + + +def test_missing_key_is_not_sent(server, configure): + configure(external_config(server)) + assert search_utils.search("q") == "Error" + assert not server.requests + + +@pytest.mark.parametrize( + "body", + [ + {}, + {"results": {}}, + {"results": [None]}, + {"results": [{"title": "x", "url": "u", "content": []}]}, + {"results": [{"title": "x", "url": "u", "content": "s", "published_date": 42}]}, + ], +) +def test_external_malformed_results(server, configure, monkeypatch, body): + configure(external_config(server)) + monkeypatch.setenv("DEEPEYES_V2_SEARCH_API_KEY", "test-credential") + server.body = body + assert search_utils.search("q") == "Error" + assert len(server.requests) == 1 + + +@pytest.mark.parametrize("status", [408, 429, 500, 503]) +def test_transient_status_is_retried(server, configure, status): + configure(retriever_config(server)) + server.status = [status, 200] + assert search_utils.search("q")["data"] == [] + assert len(server.requests) == 2 + + +@pytest.mark.parametrize("status,attempts", [(401, 1), (403, 1), (302, 1), (503, 3)]) +def test_http_failure_is_bounded_and_redacted(server, configure, caplog, status, attempts): + configure(retriever_config(server)) + server.status = status + server.body = {"detail": "sensitive-response"} + assert search_utils.search("q") == "Error" + assert len(server.requests) == attempts + assert str(status) in caplog.text + assert "sensitive-response" not in caplog.text + + +def test_read_timeout(server, configure): + configure(retriever_config(server, timeout_s=0.02, max_retries=1)) + server.delay = 0.1 + assert search_utils.search("q") == "Error" + assert len(server.requests) == 2 + + +@pytest.mark.parametrize("error", [requests.ConnectionError, requests.Timeout]) +def test_transport_retry_and_backoff(configure, monkeypatch, error): + configure({"backend": "retriever", "retriever": {"url": "http://localhost/retrieve"}}) + calls, sleeps = [], [] + + def fail(*args, **kwargs): + calls.append(kwargs) + raise error("sensitive-transport-error") + + monkeypatch.setattr(requests.Session, "request", fail) + monkeypatch.setattr(search_backends.time, "sleep", sleeps.append) + assert search_utils.search("q") == "Error" + assert len(calls) == 3 + assert sleeps == [0.5, 1.0] + assert all(c["timeout"] == 10 and c["allow_redirects"] is False for c in calls) + + +@pytest.mark.parametrize("failed", [False, True]) +async def test_agent_continues_after_search(server, configure, monkeypatch, failed): + import openai + + configure(retriever_config(server, max_retries=0)) + server.status = 503 if failed else 200 + server.body = {"result": [[{"contents": "Evidence from the retriever."}]]} + monkeypatch.setenv("OPENAI_API_KEY", "local-test") + monkeypatch.setenv("OPENAI_BASE_URL", "http://unused.invalid/v1") + monkeypatch.setattr(agent, "_build_executor", lambda *args: None) + seen = [] + + async def create(**kwargs): + seen.append(list(kwargs["messages"])) + text = ( + '{"name":"search","arguments":{"query":"q"}}' + if len(seen) == 1 + else "done" + ) + return SimpleNamespace(choices=[SimpleNamespace(message=SimpleNamespace(content=text), finish_reason="stop")]) + + monkeypatch.setattr( + openai, + "AsyncOpenAI", + lambda **kw: SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create))), + ) + result = await agent.run_session([{"role": "user", "content": "Search then answer"}], {}) + assert len(server.requests) == 1 + assert result["metadata"]["stop_reason"] == "env_done" + assert result["metadata"]["final_answer"] == "done" + assert result["metadata"]["last_error"] == ("search_failed" if failed else None) + observation = seen[1][-1] + assert observation["role"] == "tool" + assert ("Error" if failed else "Evidence from the retriever.") in observation["content"] + + +async def test_env_default_offline_and_invalid_args(monkeypatch): + env = DeepEyesV2Env(data_index="test", sandbox_executor=None, image=None) + for args in (None, {}, {"query": []}): + obs = await env.exec_tool("" + json.dumps({"name": "search", "arguments": args}) + "") + assert obs.error == "search_failed" and not obs.done + obs = await env.exec_tool('{"name":"search","arguments":{"query":"q"}}') + assert obs.error is None and not obs.done + assert "Offline mock" in obs.body_text + monkeypatch.setattr(search_utils, "_IMAGE_SEARCH_CACHE", {}) + obs = await env.exec_tool('{"name":"image_search"}') + assert obs.error == "search_failed" and not obs.done diff --git a/tests/examples/deepeyes_v2_agentic/test_search_launchers.py b/tests/examples/deepeyes_v2_agentic/test_search_launchers.py new file mode 100644 index 000000000..8639bf25c --- /dev/null +++ b/tests/examples/deepeyes_v2_agentic/test_search_launchers.py @@ -0,0 +1,83 @@ +# Copyright (c) 2026 Relax Authors. All Rights Reserved. + +"""Exercise the launch scripts with deployment and cleanup commands +replaced.""" + +from __future__ import annotations + +import json +import shlex +import shutil +import subprocess +import sys +from pathlib import Path + +import pytest + + +ROOT = Path(__file__).resolve().parents[3] + + +@pytest.mark.parametrize("suffix", ["", "_klx"]) +def test_search_settings_reach_workers_without_xtrace(tmp_path, suffix): + example = tmp_path / "examples" / "deepeyes_v2_agentic" + example.mkdir(parents=True) + filename = f"run_deepeyes_v2_agentic{suffix}.sh" + shutil.copyfile(ROOT / "examples" / "deepeyes_v2_agentic" / filename, example / filename) + (example / f"sglang_judge_service{suffix}.sh").write_text(":\n") + entrypoint = tmp_path / "scripts" / "entrypoint" + entrypoint.mkdir(parents=True) + # Use the real runtime-env builder; no cluster operations in this file. + shutil.copyfile(ROOT / "scripts" / "entrypoint" / "runtime-env-klx.sh", entrypoint / "runtime-env-klx.sh") + model_config = tmp_path / "model-config" + model_config.mkdir() + (model_config / "qwen36-35B-A3B.sh").write_text(":\n") + app_env = tmp_path / "app-env" + (app_env / ".venv" / "bin").mkdir(parents=True) + (app_env / ".venv" / "bin" / "python").symlink_to(sys.executable) + sif = tmp_path / "test.sif" + sif.touch() + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + (bin_dir / "python3").symlink_to(sys.executable) + # Prevent the launcher's startup find -exec cleanup from touching /tmp. + (bin_dir / "find").write_text("#!/bin/sh\nexit 0\n") + (bin_dir / "find").chmod(0o755) + (bin_dir / "ray").write_text( + "#!/usr/bin/env python3\nimport json, os, sys\n" + 'payload = json.loads(sys.argv[sys.argv.index("--runtime-env-json") + 1])\n' + 'with open(os.environ["CAPTURE_PATH"], "w") as f:\n json.dump(payload, f)\n' + ) + (bin_dir / "ray").chmod(0o755) + secret = 'test-key-"quoted"\\value' + config_path = str(tmp_path / 'search "quoted".yaml') + (example / "env.sh").write_text( + f"export DEEPEYES_V2_SEARCH_CONFIG={shlex.quote(config_path)}\n" + f"export DEEPEYES_V2_SEARCH_API_KEY={shlex.quote(secret)}\n" + ) + capture = tmp_path / "runtime.json" + env = { + "PATH": f"{bin_dir}:/usr/bin:/bin", + "HOME": str(tmp_path), + "WORKDIR": str(tmp_path), + "MODEL_DIR": str(tmp_path), + "DATA_DIR": str(tmp_path), + "SAVE_DIR": str(tmp_path), + "CPU_THREADS_PER_ACTOR": "4", + "MODEL_CONFIG_DIR": str(model_config), + "RELAX_ENTRYPOINT_MODE": "test", + "APPTAINER_IMAGE_PATH": str(sif), + "DEEPEYES_V2_APP_ENV_ROOT": str(app_env), + "CAPTURE_PATH": str(capture), + "PYTHONPATH": "", + } + result = subprocess.run( + ["bash", str(example / filename)], cwd=tmp_path, env=env, capture_output=True, text=True, timeout=30 + ) + assert result.returncode == 0, result.stderr + payload = json.loads(capture.read_text()) + assert payload["env_vars"]["DEEPEYES_V2_SEARCH_CONFIG"] == config_path + assert payload["env_vars"]["DEEPEYES_V2_SEARCH_API_KEY"] == secret + assert payload["env_vars"]["SANDBOX_BACKEND"] == "apptainer_jupyter" + for output in [result.stdout, result.stderr, *(p.read_text() for p in (tmp_path / "logs").glob("*.log"))]: + assert "test-key-" not in output From ba3977dc3a73cef2181e844dba46541c132e3fde Mon Sep 17 00:00:00 2001 From: WYXNICK <88577248+WYXNICK@users.noreply.github.com> Date: Sun, 20 Sep 2026 19:58:16 +0800 Subject: [PATCH 2/2] fix(deepeyes): isolate search tests from SDK MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit # 🐛 Bug Fix - Stub the optional OpenAI SDK in the scripted agent test so CPU CI can run without it. - Verify real read timeouts independently of retry request timing. --- # ♻️ Refactor - Remove repeated validation of normalized search values and redundant launcher guards. --- # ✅ Tests - All 86 DeepEyes example tests pass without the OpenAI SDK installed. - Full pre-commit checks and launcher syntax checks pass. --- .../app/search_backends.py | 4 ++-- .../run_deepeyes_v2_agentic.sh | 2 +- .../run_deepeyes_v2_agentic_klx.sh | 3 +-- .../test_search_backends.py | 23 +++++++++++-------- 4 files changed, 18 insertions(+), 14 deletions(-) diff --git a/examples/deepeyes_v2_agentic/app/search_backends.py b/examples/deepeyes_v2_agentic/app/search_backends.py index a96e1eb3a..ca2425182 100644 --- a/examples/deepeyes_v2_agentic/app/search_backends.py +++ b/examples/deepeyes_v2_agentic/app/search_backends.py @@ -144,7 +144,7 @@ def _retriever_rows(body: Any, size: int) -> list[dict]: link = doc.get("link", "") rows.append( { - "title": _text(title, "result.title", empty=True), + "title": title, "link": _text("" if link is None else link, "result.link", empty=True), "snippet": content, "date": _date(doc.get("date")), @@ -209,7 +209,7 @@ def run_search(query: str, size: int | None = None) -> dict: """Return normalized search results, or raise SearchError on failure.""" query = _text(query, "query").strip() config = _load_config() - size = _integer(config["top_k"] if size is None else size, "size", 1) + size = config["top_k"] if size is None else _integer(size, "size", 1) if config["backend"] == "mock": return { "elapsed_time": 0.0, diff --git a/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic.sh b/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic.sh index 5a79ef599..18c30d782 100755 --- a/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic.sh +++ b/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic.sh @@ -277,7 +277,7 @@ payload = json.loads(os.environ["RUNTIME_ENV_JSON"]) for name in ("DEEPEYES_V2_SEARCH_CONFIG", "DEEPEYES_V2_SEARCH_API_KEY"): value = os.environ.get(name) if value: - payload.setdefault("env_vars", {})[name] = value + payload["env_vars"][name] = value print(json.dumps(payload)) PYTHON ) diff --git a/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic_klx.sh b/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic_klx.sh index e74642ef1..632661df2 100644 --- a/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic_klx.sh +++ b/examples/deepeyes_v2_agentic/run_deepeyes_v2_agentic_klx.sh @@ -337,7 +337,6 @@ PY ) fi # Pass search configuration to Ray workers without tracing credentials. -set +x export RUNTIME_ENV_JSON RUNTIME_ENV_JSON=$(python3 - <<'PYTHON' import json @@ -347,7 +346,7 @@ payload = json.loads(os.environ["RUNTIME_ENV_JSON"]) for name in ("DEEPEYES_V2_SEARCH_CONFIG", "DEEPEYES_V2_SEARCH_API_KEY"): value = os.environ.get(name) if value: - payload.setdefault("env_vars", {})[name] = value + payload["env_vars"][name] = value print(json.dumps(payload)) PYTHON ) diff --git a/tests/examples/deepeyes_v2_agentic/test_search_backends.py b/tests/examples/deepeyes_v2_agentic/test_search_backends.py index 0f21a93c0..bac8d0383 100644 --- a/tests/examples/deepeyes_v2_agentic/test_search_backends.py +++ b/tests/examples/deepeyes_v2_agentic/test_search_backends.py @@ -321,11 +321,11 @@ def test_http_failure_is_bounded_and_redacted(server, configure, caplog, status, assert "sensitive-response" not in caplog.text -def test_read_timeout(server, configure): - configure(retriever_config(server, timeout_s=0.02, max_retries=1)) - server.delay = 0.1 +def test_read_timeout(server, configure, caplog): + configure(retriever_config(server, timeout_s=0.05, max_retries=0)) + server.delay = 0.2 assert search_utils.search("q") == "Error" - assert len(server.requests) == 2 + assert "ReadTimeout" in caplog.text @pytest.mark.parametrize("error", [requests.ConnectionError, requests.Timeout]) @@ -347,7 +347,8 @@ def fail(*args, **kwargs): @pytest.mark.parametrize("failed", [False, True]) async def test_agent_continues_after_search(server, configure, monkeypatch, failed): - import openai + class APIStatusError(Exception): + pass configure(retriever_config(server, max_retries=0)) server.status = 503 if failed else 200 @@ -366,10 +367,14 @@ async def create(**kwargs): ) return SimpleNamespace(choices=[SimpleNamespace(message=SimpleNamespace(content=text), finish_reason="stop")]) - monkeypatch.setattr( - openai, - "AsyncOpenAI", - lambda **kw: SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create))), + # The scripted model does not require the optional OpenAI SDK. + monkeypatch.setitem( + sys.modules, + "openai", + SimpleNamespace( + APIStatusError=APIStatusError, + AsyncOpenAI=lambda **kw: SimpleNamespace(chat=SimpleNamespace(completions=SimpleNamespace(create=create))), + ), ) result = await agent.run_session([{"role": "user", "content": "Search then answer"}], {}) assert len(server.requests) == 1