|
| 1 | +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. |
| 2 | +# |
| 3 | +# Licensed under the Apache License, Version 2.0 (the "License"); |
| 4 | +# you may not use this file except in compliance with the License. |
| 5 | +# You may obtain a copy of the License at |
| 6 | +# |
| 7 | +# http://www.apache.org/licenses/LICENSE-2.0 |
| 8 | +# |
| 9 | +# Unless required by applicable law or agreed to in writing, software |
| 10 | +# distributed under the License is distributed on an "AS IS" BASIS, |
| 11 | +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. |
| 12 | +# See the License for the specific language governing permissions and |
| 13 | +# limitations under the License. |
| 14 | + |
| 15 | +"""Smoke tests against the real TypeSafe System One endpoint. |
| 16 | +
|
| 17 | +TypeSafe is a paid, shared service, so these tests are opt in and stay out of a |
| 18 | +normal run: |
| 19 | +
|
| 20 | +```bash |
| 21 | +TYPESAFE_RUN_SMOKE=1 TYPESAFE_API_KEY=... pytest -m typesafe_smoke |
| 22 | +``` |
| 23 | +
|
| 24 | +They answer one question the offline tests cannot: whether the contract the |
| 25 | +plugin was written against still holds on the live endpoint. |
| 26 | +""" |
| 27 | + |
| 28 | +from __future__ import annotations |
| 29 | + |
| 30 | +import asyncio |
| 31 | +import os |
| 32 | + |
| 33 | +import pytest |
| 34 | + |
| 35 | +from veadk.extensions.decisions import ( |
| 36 | + ChoiceAnswer, |
| 37 | + DecisionExtension, |
| 38 | + DecisionModelConfig, |
| 39 | + NoulAnswer, |
| 40 | + ScoreAnswer, |
| 41 | + choice_question, |
| 42 | + noul_question, |
| 43 | + score_question, |
| 44 | +) |
| 45 | +from veadk.extensions.harness.modules.agent_routing import DecisionAgentRouter |
| 46 | +from veadk.extensions.harness.modules.final_response_verifier.support_judge import ( |
| 47 | + DecisionSupportJudge, |
| 48 | +) |
| 49 | +from veadk.extensions.harness.modules.invocation_context.mode_judge import ( |
| 50 | + DecisionModeJudge, |
| 51 | +) |
| 52 | +from veadk.extensions.harness.modules.long_run_control.judge import ( |
| 53 | + DecisionConvergenceJudge, |
| 54 | +) |
| 55 | +from veadk.extensions.harness.modules.skill_prefilter import DecisionSkillJudge |
| 56 | +from veadk.extensions.harness.modules.tool_result_compactor import ( |
| 57 | + DecisionCompactionJudge, |
| 58 | +) |
| 59 | +from veadk.extensions.harness.schemas import ToolReceipt |
| 60 | +from veadk.memory.auto_save_judge import DecisionMemorySaveJudge |
| 61 | +from veadk.memory.recall_judge import DecisionRecallJudge |
| 62 | + |
| 63 | +API_KEY = os.environ.get("TYPESAFE_API_KEY", "") |
| 64 | +API_BASE = os.environ.get("TYPESAFE_BASE_URL", "https://api.typesafe.ai") |
| 65 | +RUN_SMOKE = os.environ.get("TYPESAFE_RUN_SMOKE") == "1" |
| 66 | + |
| 67 | +pytestmark = [ |
| 68 | + pytest.mark.typesafe_smoke, |
| 69 | + pytest.mark.skipif( |
| 70 | + not (RUN_SMOKE and API_KEY), |
| 71 | + reason="set TYPESAFE_RUN_SMOKE=1 and TYPESAFE_API_KEY to call the live endpoint", |
| 72 | + ), |
| 73 | +] |
| 74 | + |
| 75 | +URGENT_TICKET = ( |
| 76 | + "Hi, I have been trying to connect my Stripe account for 3 days and the " |
| 77 | + "integration keeps failing. I am losing sales. Please help ASAP." |
| 78 | +) |
| 79 | + |
| 80 | + |
| 81 | +def _extension() -> DecisionExtension: |
| 82 | + return DecisionExtension( |
| 83 | + DecisionModelConfig(enabled=True, api_base=API_BASE, api_key=API_KEY) |
| 84 | + ) |
| 85 | + |
| 86 | + |
| 87 | +def test_one_call_answers_all_three_primitives() -> None: |
| 88 | + result = _extension().evaluate( |
| 89 | + state=URGENT_TICKET, |
| 90 | + questions={ |
| 91 | + "is_urgent": noul_question("Does this message express urgency?"), |
| 92 | + "department": choice_question( |
| 93 | + "Which team should handle this", |
| 94 | + {"billing": "Payment issues", "technical": "Integration problems"}, |
| 95 | + ), |
| 96 | + "frustration": score_question( |
| 97 | + "How frustrated is the customer?", ["Calm", "Frustrated", "Angry"] |
| 98 | + ), |
| 99 | + }, |
| 100 | + ) |
| 101 | + |
| 102 | + assert result.model, "the response must name the model that answered" |
| 103 | + urgent = result.answers["is_urgent"] |
| 104 | + assert isinstance(urgent, NoulAnswer) and 0.0 <= urgent.noul <= 1.0 |
| 105 | + |
| 106 | + department = result.answers["department"] |
| 107 | + assert isinstance(department, ChoiceAnswer) |
| 108 | + assert department.choice in {"billing", "technical"} |
| 109 | + assert set(department.probabilities) == {"billing", "technical"} |
| 110 | + assert sum(department.probabilities.values()) == pytest.approx(1.0, abs=0.01) |
| 111 | + assert 0.0 <= department.confidence <= 1.0 |
| 112 | + |
| 113 | + frustration = result.answers["frustration"] |
| 114 | + assert isinstance(frustration, ScoreAnswer) |
| 115 | + assert set(frustration.legend) == {"0", "1", "2"} |
| 116 | + assert 0.0 <= frustration.score <= 2.0 |
| 117 | + assert sum(frustration.probabilities.values()) == pytest.approx(1.0, abs=0.01) |
| 118 | + assert 0.0 <= frustration.confidence <= 1.0 |
| 119 | + |
| 120 | + |
| 121 | +def test_environment_configuration_reaches_the_live_endpoint( |
| 122 | + monkeypatch: pytest.MonkeyPatch, |
| 123 | +) -> None: |
| 124 | + """What ``from_env`` builds must be usable against the real endpoint.""" |
| 125 | + monkeypatch.setenv("DECISION_MODEL_ENABLED", "true") |
| 126 | + monkeypatch.setenv("DECISION_MODEL_API_KEY", API_KEY) |
| 127 | + monkeypatch.setenv("DECISION_MODEL_API_BASE", API_BASE) |
| 128 | + |
| 129 | + extension = DecisionExtension(DecisionModelConfig.from_env()) |
| 130 | + |
| 131 | + assert extension.enabled |
| 132 | + result = asyncio.run( |
| 133 | + extension.aevaluate( |
| 134 | + state=URGENT_TICKET, |
| 135 | + questions={ |
| 136 | + "is_urgent": noul_question("Does this message express urgency?") |
| 137 | + }, |
| 138 | + ) |
| 139 | + ) |
| 140 | + assert isinstance(result.answers["is_urgent"], NoulAnswer) |
| 141 | + |
| 142 | + |
| 143 | +def test_every_judgement_point_answers_against_the_live_endpoint() -> None: |
| 144 | + """Each judgement point must come back with a usable, in-range answer.""" |
| 145 | + |
| 146 | + async def judge_all() -> None: |
| 147 | + extension = _extension() |
| 148 | + |
| 149 | + routed = await DecisionAgentRouter(extension).aroute( |
| 150 | + user_input=URGENT_TICKET, |
| 151 | + agents={ |
| 152 | + "billing_agent": "handles invoices and refunds", |
| 153 | + "docs_agent": "answers product questions", |
| 154 | + }, |
| 155 | + ) |
| 156 | + assert routed in {"billing_agent", "docs_agent", None} |
| 157 | + |
| 158 | + skills = await DecisionSkillJudge(extension).aprobabilities( |
| 159 | + user_input="Draw a sequence diagram for the checkout flow.", |
| 160 | + skills={ |
| 161 | + "archify": "renders architecture and sequence diagrams", |
| 162 | + "pptx": "builds slide decks", |
| 163 | + }, |
| 164 | + ) |
| 165 | + assert set(skills) == {"archify", "pptx"} |
| 166 | + assert all(0.0 <= value <= 1.0 for value in skills.values()) |
| 167 | + |
| 168 | + kept = await DecisionCompactionJudge(extension).aprotect( |
| 169 | + goal="rank the candidates by score", |
| 170 | + evidence={ |
| 171 | + 1: "candidate A scored 0.91 with 3 matching skills", |
| 172 | + 2: "small talk about lunch", |
| 173 | + }, |
| 174 | + ) |
| 175 | + assert set(kept) == {1, 2} |
| 176 | + assert all(0.0 <= value <= 1.0 for value in kept.values()) |
| 177 | + |
| 178 | + judgement = await DecisionSupportJudge(extension).areview( |
| 179 | + answer="Done, I deployed the service and it is healthy.", |
| 180 | + receipts=[ |
| 181 | + ToolReceipt( |
| 182 | + name="run_shell", status="success", summary="wrote notes.md" |
| 183 | + ) |
| 184 | + ], |
| 185 | + goal="deploy the service", |
| 186 | + ) |
| 187 | + assert judgement.verdict in {"supported", "partial", "unsupported"} |
| 188 | + assert 0.0 <= judgement.support <= 1.0 |
| 189 | + assert 0.0 <= judgement.confidence <= 1.0 |
| 190 | + |
| 191 | + modes = await DecisionModeJudge(extension).aprobabilities( |
| 192 | + user_input="Refactor the parser and run the tests." |
| 193 | + ) |
| 194 | + assert all(0.0 <= value <= 1.0 for value in modes.values()) |
| 195 | + |
| 196 | + convergence = await DecisionConvergenceJudge(extension).ajudge( |
| 197 | + goal="make the failing test pass", |
| 198 | + trajectory=( |
| 199 | + "step 1: ran pytest, 3 failures\n" |
| 200 | + "step 2: fixed the fixture, 1 failure left\n" |
| 201 | + "step 3: ran pytest again, 1 failure remains" |
| 202 | + ), |
| 203 | + ) |
| 204 | + assert 0.0 <= convergence.ready <= 1.0 |
| 205 | + |
| 206 | + relevance = await DecisionRecallJudge(extension).arelevance( |
| 207 | + query="what does the user prefer for diagrams?", |
| 208 | + memories=[ |
| 209 | + "the user prefers diagrams over prose", |
| 210 | + "the user lives in Shanghai", |
| 211 | + ], |
| 212 | + ) |
| 213 | + assert set(relevance) == {0, 1} |
| 214 | + assert all(0.0 <= value <= 1.0 for value in relevance.values()) |
| 215 | + assert relevance[0] > relevance[1] |
| 216 | + |
| 217 | + worth_saving = await DecisionMemorySaveJudge(extension).aworth_saving( |
| 218 | + events_text="user: my preferred timezone is Asia/Shanghai; agent: noted." |
| 219 | + ) |
| 220 | + assert 0.0 <= worth_saving <= 1.0 |
| 221 | + |
| 222 | + asyncio.run(judge_all()) |
0 commit comments