Skip to content

Commit 56b4e7d

Browse files
committed
feat(decisions): degrade gracefully when the decision model is unavailable
The decision model sits in hot paths, so a failure has to reach callers as a ``DecisionModelError`` they can fall back from, and a down endpoint may not keep costing the retry and timeout budget: - Normalize every failure: a send error that is not an ``httpx.HTTPError`` (for example ``httpx.InvalidURL``) and a payload that fails pydantic validation now surface as ``DecisionModelRequestError`` / ``DecisionModelResponseError`` instead of escaping past the callers' ``except DecisionModelError``. - ``timeout`` is the budget of one whole judgement, retries and backoff included, and is capped at 5 seconds. - After ``failure_threshold`` failures in a row the endpoint is marked down for ``cooldown_seconds``; judgements then raise ``DecisionModelUnavailableError`` without an HTTP call, and one probe after the cooldown decides whether to resume. - Log one line per usable judgement at debug, retries and outages at warning; never log the judged state or the API key. - Reject an unusable ``api_base`` at configuration time, and keep the extension disabled instead of failing startup when settings are broken. Change-Id: I58b57993c4791c6e1c57e7c52af85af53bd05c85
1 parent abb2f15 commit 56b4e7d

11 files changed

Lines changed: 704 additions & 65 deletions

File tree

‎tests/extensions/decisions/test_client.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -146,7 +146,7 @@ def test_rate_limit_is_retried_and_can_succeed() -> None:
146146
def test_retries_are_bounded() -> None:
147147
script = [(529, {"retry-after": "0"}, {"detail": "overloaded"})] * 3
148148
with fake_system_one(script) as server:
149-
with pytest.raises(DecisionModelRequestError, match="after 2 retries"):
149+
with pytest.raises(DecisionModelRequestError, match="after 3 attempt"):
150150
_client(server.base_url, max_retries=2).evaluate(
151151
state="hi", questions={"q": noul_question("Is this a greeting?")}
152152
)

‎tests/extensions/decisions/test_config.py‎

Lines changed: 43 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -21,7 +21,10 @@
2121

2222
from veadk.extensions.decisions import (
2323
DEFAULT_API_BASE,
24+
DEFAULT_COOLDOWN_SECONDS,
25+
DEFAULT_FAILURE_THRESHOLD,
2426
DEFAULT_MODEL_NAME,
27+
MAX_TIMEOUT_SECONDS,
2528
OPENROUTER_API_BASE,
2629
DecisionModelConfig,
2730
)
@@ -44,17 +47,21 @@ def test_from_env_reads_every_field() -> None:
4447
"DECISION_MODEL_NAME": "jev-1.13.0",
4548
"DECISION_MODEL_API_BASE": "http://localhost:9000/",
4649
"DECISION_MODEL_API_KEY": "secret",
47-
"DECISION_MODEL_TIMEOUT": "12.5",
50+
"DECISION_MODEL_TIMEOUT": "4.5",
4851
"DECISION_MODEL_MAX_RETRIES": "1",
52+
"DECISION_MODEL_FAILURE_THRESHOLD": "5",
53+
"DECISION_MODEL_COOLDOWN_SECONDS": "12.5",
4954
}
5055
)
5156
assert config.enabled is True
5257
assert config.provider == "systemone"
5358
assert config.name == "jev-1.13.0"
5459
assert config.api_base == "http://localhost:9000"
5560
assert config.api_key == "secret"
56-
assert config.timeout == 12.5
61+
assert config.timeout == 4.5
5762
assert config.max_retries == 1
63+
assert config.failure_threshold == 5
64+
assert config.cooldown_seconds == 12.5
5865
assert config.endpoint == "http://localhost:9000/v1/systemone"
5966
assert config.configured is True
6067

@@ -70,8 +77,41 @@ def test_from_env_keeps_defaults_for_unusable_values() -> None:
7077
)
7178
assert config.enabled is False
7279
assert config.provider == "typesafe"
73-
assert config.timeout == 30.0
80+
assert config.timeout == MAX_TIMEOUT_SECONDS
7481
assert config.max_retries == 3
82+
assert config.failure_threshold == DEFAULT_FAILURE_THRESHOLD
83+
assert config.cooldown_seconds == DEFAULT_COOLDOWN_SECONDS
84+
85+
86+
def test_timeout_is_capped_so_one_judgement_cannot_stall_a_run() -> None:
87+
"""A judgement sits before and after model calls, so it has a hard ceiling."""
88+
assert DecisionModelConfig().timeout == MAX_TIMEOUT_SECONDS
89+
assert DecisionModelConfig(timeout=30.0).timeout == MAX_TIMEOUT_SECONDS
90+
assert (
91+
DecisionModelConfig.from_env({"DECISION_MODEL_TIMEOUT": "300"}).timeout
92+
== MAX_TIMEOUT_SECONDS
93+
)
94+
95+
96+
@pytest.mark.parametrize(
97+
"api_base",
98+
["http://[::1", "not a url", "ftp://host", "http://", "https://user:pw@host"],
99+
)
100+
def test_unusable_api_base_is_rejected_at_configuration_time(api_base: str) -> None:
101+
with pytest.raises(ValidationError):
102+
DecisionModelConfig(api_base=api_base)
103+
104+
105+
def test_from_env_disables_itself_instead_of_breaking_startup() -> None:
106+
"""Broken settings degrade to "no decision model" with a warning."""
107+
config = DecisionModelConfig.from_env(
108+
{
109+
"DECISION_MODEL_ENABLED": "true",
110+
"DECISION_MODEL_API_KEY": "secret",
111+
"DECISION_MODEL_API_BASE": "http://[::1",
112+
}
113+
)
114+
assert config.configured is False
75115

76116

77117
@pytest.mark.parametrize(
Lines changed: 238 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,238 @@
1+
# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates.
2+
#
3+
# Licensed under the Apache License, Version 2.0 (the "License");
4+
# you may not use this file except in compliance with the License.
5+
# You may obtain a copy of the License at
6+
#
7+
# http://www.apache.org/licenses/LICENSE-2.0
8+
#
9+
# Unless required by applicable law or agreed to in writing, software
10+
# distributed under the License is distributed on an "AS IS" BASIS,
11+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12+
# See the License for the specific language governing permissions and
13+
# limitations under the License.
14+
15+
"""Failure handling, degradation, and availability tests.
16+
17+
The decision model is optional and sits in hot paths, so every failure has to
18+
reach callers as a ``DecisionModelError`` they can fall back from, and a
19+
down endpoint must stop costing the retry and timeout budget.
20+
"""
21+
22+
from __future__ import annotations
23+
24+
import logging
25+
import time
26+
27+
import httpx
28+
import pytest
29+
30+
from veadk.extensions.decisions import (
31+
DecisionExtension,
32+
DecisionModelConfig,
33+
DecisionModelRequestError,
34+
DecisionModelResponseError,
35+
DecisionModelUnavailableError,
36+
SystemOneClient,
37+
noul_question,
38+
)
39+
from veadk.extensions.decisions.types import parse_answers
40+
41+
from .fake_system_one import FakeSystemOneServer, fake_system_one
42+
43+
QUESTION = {"q": noul_question("Is this a greeting?")}
44+
45+
46+
def _config(
47+
server_url: str,
48+
*,
49+
timeout: float = 5.0,
50+
max_retries: int = 0,
51+
failure_threshold: int = 3,
52+
cooldown_seconds: float = 30.0,
53+
) -> DecisionModelConfig:
54+
return DecisionModelConfig(
55+
enabled=True,
56+
api_base=server_url,
57+
api_key="test-key",
58+
name="jev-latest",
59+
timeout=timeout,
60+
max_retries=max_retries,
61+
failure_threshold=failure_threshold,
62+
cooldown_seconds=cooldown_seconds,
63+
)
64+
65+
66+
def _extension(server_url: str, **settings: float | int) -> DecisionExtension:
67+
return DecisionExtension(_config(server_url, **settings))
68+
69+
70+
def _schedule_outage(server: FakeSystemOneServer, count: int = 1) -> None:
71+
"""Make the next ``count`` requests fail like a down endpoint."""
72+
server.script.extend([(503, {}, {"detail": "unavailable"})] * count)
73+
74+
75+
def _judge(extension: DecisionExtension) -> float:
76+
"""Ask one noul question and return the probability."""
77+
return extension.noul("state", "Is this a greeting?").noul
78+
79+
80+
# -- every failure is a decision-model error -------------------------------
81+
82+
83+
def test_a_malformed_answer_is_a_response_error() -> None:
84+
"""A payload that fails pydantic validation must not escape as itself."""
85+
with pytest.raises(DecisionModelResponseError, match="not a valid choice"):
86+
parse_answers({"q": {"type": "choice"}})
87+
with pytest.raises(DecisionModelResponseError, match="not a valid noul"):
88+
parse_answers({"q": {"type": "noul", "noul": "certainly"}})
89+
90+
91+
def test_a_broken_answer_payload_reaches_callers_as_a_decision_error() -> None:
92+
script = [(200, {}, {"model": "fake", "answers": {"q": {"type": "choice"}}})]
93+
with fake_system_one(script) as server:
94+
with pytest.raises(DecisionModelResponseError):
95+
_judge(_extension(server.base_url))
96+
97+
98+
def test_a_send_failure_that_is_not_an_http_error_is_normalized(
99+
monkeypatch: pytest.MonkeyPatch,
100+
) -> None:
101+
"""``httpx.InvalidURL`` is not an ``httpx.HTTPError``, so it needs our own."""
102+
assert not issubclass(httpx.InvalidURL, httpx.HTTPError)
103+
104+
def _raise_invalid_url(*_args: object, **_kwargs: object) -> None:
105+
raise httpx.InvalidURL("Invalid port: ':1'")
106+
107+
monkeypatch.setattr(httpx.Client, "post", _raise_invalid_url)
108+
client = SystemOneClient(_config("https://api.example.com"))
109+
with pytest.raises(DecisionModelRequestError, match="could not be sent"):
110+
client.evaluate(state="state", questions=QUESTION)
111+
112+
113+
def test_a_transport_outage_is_a_request_error() -> None:
114+
with fake_system_one() as server:
115+
base_url = server.base_url
116+
with pytest.raises(DecisionModelRequestError):
117+
SystemOneClient(_config(base_url)).evaluate(state="state", questions=QUESTION)
118+
119+
120+
# -- the endpoint stops being called once it is known to be down -----------
121+
122+
123+
def test_repeated_failures_open_the_circuit_and_skip_the_endpoint() -> None:
124+
with fake_system_one() as server:
125+
extension = _extension(server.base_url, failure_threshold=2)
126+
_schedule_outage(server, 2)
127+
for _ in range(2):
128+
with pytest.raises(DecisionModelRequestError):
129+
_judge(extension)
130+
assert len(server.calls) == 2
131+
132+
with pytest.raises(DecisionModelUnavailableError, match="marked down"):
133+
_judge(extension)
134+
assert len(server.calls) == 2
135+
136+
137+
def test_a_successful_probe_resumes_judgements() -> None:
138+
with fake_system_one() as server:
139+
extension = _extension(
140+
server.base_url, failure_threshold=1, cooldown_seconds=0.05
141+
)
142+
_schedule_outage(server)
143+
with pytest.raises(DecisionModelRequestError):
144+
_judge(extension)
145+
with pytest.raises(DecisionModelUnavailableError):
146+
_judge(extension)
147+
148+
time.sleep(0.06)
149+
assert _judge(extension) == pytest.approx(0.9)
150+
assert _judge(extension) == pytest.approx(0.9)
151+
152+
153+
def test_a_failed_probe_keeps_the_circuit_open() -> None:
154+
with fake_system_one() as server:
155+
extension = _extension(
156+
server.base_url, failure_threshold=1, cooldown_seconds=0.05
157+
)
158+
_schedule_outage(server)
159+
with pytest.raises(DecisionModelRequestError):
160+
_judge(extension)
161+
162+
time.sleep(0.06)
163+
_schedule_outage(server)
164+
with pytest.raises(DecisionModelRequestError):
165+
_judge(extension)
166+
with pytest.raises(DecisionModelUnavailableError):
167+
_judge(extension)
168+
# 三次失败里只有两次真的打了上游:冷却期内的那次被拦下了
169+
assert len(server.calls) == 2
170+
171+
172+
def test_the_circuit_closes_after_isolated_failures() -> None:
173+
with fake_system_one() as server:
174+
extension = _extension(server.base_url, failure_threshold=2)
175+
_schedule_outage(server)
176+
with pytest.raises(DecisionModelRequestError):
177+
_judge(extension)
178+
assert _judge(extension) == pytest.approx(0.9)
179+
180+
_schedule_outage(server)
181+
with pytest.raises(DecisionModelRequestError):
182+
_judge(extension)
183+
assert _judge(extension) == pytest.approx(0.9)
184+
185+
186+
def test_the_circuit_can_be_disabled() -> None:
187+
with fake_system_one() as server:
188+
extension = _extension(server.base_url, failure_threshold=0)
189+
_schedule_outage(server, 3)
190+
for _ in range(3):
191+
with pytest.raises(DecisionModelRequestError):
192+
_judge(extension)
193+
assert len(server.calls) == 3
194+
195+
196+
# -- the timeout is a budget for the whole judgement -----------------------
197+
198+
199+
def test_one_judgement_stays_within_its_time_budget() -> None:
200+
"""Retries and backoff cannot outlive the configured budget."""
201+
script = [(503, {"retry-after": "5"}, {"detail": "unavailable"})] * 4
202+
with fake_system_one(script) as server:
203+
client = SystemOneClient(_config(server.base_url, timeout=0.3, max_retries=3))
204+
started = time.perf_counter()
205+
with pytest.raises(DecisionModelRequestError, match="time budget"):
206+
client.evaluate(state="state", questions=QUESTION)
207+
elapsed = time.perf_counter() - started
208+
assert elapsed < 0.6
209+
assert len(server.calls) == 1
210+
211+
212+
# -- what ends up in the log ----------------------------------------------
213+
214+
215+
def test_retries_and_outages_are_logged(caplog: pytest.LogCaptureFixture) -> None:
216+
script = [(503, {"retry-after": "0"}, {"detail": "unavailable"})] * 2
217+
with fake_system_one(script) as server:
218+
extension = _extension(server.base_url, max_retries=1, failure_threshold=1)
219+
with caplog.at_level(logging.WARNING):
220+
with pytest.raises(DecisionModelRequestError):
221+
_judge(extension)
222+
223+
messages = [record.getMessage() for record in caplog.records]
224+
assert any("retrying" in message for message in messages)
225+
assert any("in a row" in message for message in messages)
226+
assert all("test-key" not in message for message in messages)
227+
228+
229+
def test_the_judged_state_is_never_logged(caplog: pytest.LogCaptureFixture) -> None:
230+
with fake_system_one() as server:
231+
extension = _extension(server.base_url)
232+
with caplog.at_level(logging.DEBUG):
233+
extension.noul("SECRET-USER-TEXT", "Is this a greeting?")
234+
235+
assert caplog.records, "a successful judgement should be observable"
236+
assert all(
237+
"SECRET-USER-TEXT" not in record.getMessage() for record in caplog.records
238+
)

‎veadk/extensions/decisions/README.md‎

Lines changed: 34 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -28,10 +28,17 @@ DECISION_MODEL_PROVIDER=typesafe # typesafe | openrouter | systemone
2828
DECISION_MODEL_NAME=jev-latest
2929
DECISION_MODEL_API_BASE=https://api.typesafe.ai
3030
DECISION_MODEL_API_KEY=...
31-
DECISION_MODEL_TIMEOUT=30
31+
DECISION_MODEL_TIMEOUT=5 # seconds per judgement (ceiling: 5)
3232
DECISION_MODEL_MAX_RETRIES=3
33+
DECISION_MODEL_FAILURE_THRESHOLD=3 # 0 disables the circuit breaker
34+
DECISION_MODEL_COOLDOWN_SECONDS=30
3335
```
3436

37+
`DECISION_MODEL_TIMEOUT` is the budget of one whole judgement, retries and
38+
their backoff included. Values above 5 seconds are clamped to 5 with a
39+
warning: a judgement runs in the agent's hot path, so a slow endpoint has to
40+
degrade the judgement rather than the run.
41+
3542
Every provider speaks the same System One protocol, so switching only changes
3643
the API base and the API key. The provider picks the default `api_base`:
3744

@@ -103,6 +110,32 @@ The tool asks the configured decision model for one judgement and returns
103110
decision model is unconfigured or the request fails, so a run never breaks
104111
because of an optional capability.
105112

113+
## Failures and Degradation
114+
115+
The decision model is optional, so callers only ever handle one error type:
116+
`DecisionModelError`. Whatever goes wrong, a judgement degrades to the caller's
117+
own rules instead of breaking the run.
118+
119+
| Failure | What happens |
120+
| --- | --- |
121+
| Not configured, or no API key | `DecisionModelDisabledError` on the first call; nothing else changes. |
122+
| Timeout, connection error, `429`, `5xx` | Retried with exponential backoff inside the `timeout` budget, honouring `retry-after`; then `DecisionModelRequestError`. |
123+
| Other `4xx` | Not retried; `DecisionModelRequestError` with the status and a body snippet. |
124+
| `200` with unusable answers | `DecisionModelResponseError`; malformed payloads and unknown answer types are reported the same way, never as a `pydantic` or `httpx` error. |
125+
| `DECISION_MODEL_TIMEOUT` above the ceiling | Clamped to 5 seconds with a warning. |
126+
| Unusable settings at startup | The extension disables itself with a warning instead of failing startup. |
127+
| `DECISION_MODEL_FAILURE_THRESHOLD` failures in a row (default 3) | The endpoint is marked down for `DECISION_MODEL_COOLDOWN_SECONDS` (default 30). Further judgements raise `DecisionModelUnavailableError` immediately and make no HTTP call; one probe request after the cooldown decides whether to resume. |
128+
129+
Logging stays quiet and carries no user data:
130+
131+
| Level | Message |
132+
| --- | --- |
133+
| `DEBUG` | One line per usable judgement: model, latency, tokens, cost. |
134+
| `INFO` | Cooldown elapsed and a probe was sent; judgements resumed. |
135+
| `WARNING` | A retry, an outage that marked the endpoint down, settings that were clamped or are unusable. |
136+
137+
The judged state and the API key are never logged.
138+
106139
## Source Layout
107140

108141
| Path | Purpose |

0 commit comments

Comments
 (0)