diff --git a/.github/banner.html b/.github/banner.html index cc9b8f2..aff03ae 100644 --- a/.github/banner.html +++ b/.github/banner.html @@ -130,7 +130,7 @@ hotato
Self-hosted  ·  MIT
-

Conversation QA for voice agents

+

Testing and observability for AI agents

See exactly why a call passed, or failed. Five dimensions, scored separately, each with the evidence behind it.

$uvx hotato start --demo diff --git a/.github/social-card.html b/.github/social-card.html index a7c8c37..c3db55c 100644 --- a/.github/social-card.html +++ b/.github/social-card.html @@ -93,7 +93,7 @@ hotato
-

Conversation QA for voice agents

+

Testing and observability for AI agents

See exactly why a call passed, or failed, with the evidence behind every verdict.

diff --git a/.gitignore b/.gitignore index 45c9976..f4076d1 100644 --- a/.gitignore +++ b/.gitignore @@ -102,3 +102,6 @@ docs/SAA-BEHAVIOR-CARD.md docs/SAA-FIX-POINTER.md docs/saa_behavior_card.py tests/test_saa_card.py + +# hotato start writes its demo saydo/ output to CWD; ignore it at repo root +/saydo/ diff --git a/MANIFEST.in b/MANIFEST.in index 5a55bb0..3f6e12b 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -107,3 +107,10 @@ exclude docs/saa_behavior_card.py exclude tests/test_saa_card.py recursive-exclude docs SAA-*.md saa_*.py recursive-exclude tests test_saa_*.py + +# Rendered showcase reports with base64-embedded audio (~12 MB combined) are +# documentation artifacts, not runtime dependencies: no code or test reads them. +# Keep them in git (linked from the corpus READMEs) but exclude them from the +# shipped sdist so a plain `pip install hotato` is ~0.6 MB instead of ~12 MB. +exclude corpus/vapi-defaults/sample-report.html +exclude corpus/real/sample-report.html diff --git a/docs/CONTRACTS.md b/docs/CONTRACTS.md index 0412578..0805c68 100644 --- a/docs/CONTRACTS.md +++ b/docs/CONTRACTS.md @@ -28,7 +28,7 @@ reports pass/fail. | **Re-scores** | the SAME `audio/event.wav` the contract was created from | a new recording of the SAME stimulus against your CURRENT agent | | **A pass proves** | the evidence, policy, and scorer are still intact and still agree with the human label | the CURRENT agent's behavior on that stimulus still matches the label | | **Leaves open** | whether the deployed agent's behavior has changed since | nothing extra -- this lane speaks to the live agent | -| **Runs** | every push, in the shipped `ci/github-action.yml` (`contract verify contracts/`) | only when you recapture by hand or on a schedule -- see [`docs/RECAPTURE.md`](RECAPTURE.md) | +| **Runs** | every push, in the shipped `ci/github_action.yml` (`contract verify contracts/`) | only when you recapture by hand or on a schedule -- see [`docs/RECAPTURE.md`](RECAPTURE.md) | A frozen-recording pass is necessary but not sufficient: the recording never changes, so it can only fail if someone edits the bundle's audio @@ -149,7 +149,7 @@ Exit codes are the CI contract: | `2` | usage error, empty directory, or corrupt `contract.json` | `--junit` writes one `` per contract; the shipped -`ci/github-action.yml` scaffold runs this on push, on PR, and weekly, +`ci/github_action.yml` scaffold runs this on push, on PR, and weekly, and publishes the JUnit file as an artifact. Every text and HTML render of `verify` also prints, verbatim: *"This @@ -238,7 +238,7 @@ matching the packed manifest. ## CI -The shipped `ci/github-action.yml` is the minimal wiring: +The shipped `ci/github_action.yml` is the minimal wiring: ```bash uvx hotato contract verify contracts/ --junit contracts-junit.xml \ diff --git a/docs/LIFECYCLE.md b/docs/LIFECYCLE.md index 48b081c..4f99890 100644 --- a/docs/LIFECYCLE.md +++ b/docs/LIFECYCLE.md @@ -91,9 +91,13 @@ deterministic: the same scenario and seed render the same bytes on every run. ## 5. Prove -One command composes every evidence lane you have into one fail-closed release -proof: contracts re-verified, suites re-run, before/after movement measured, -the stress suite cleared. The proof is a content-addressed receipt; CI gates on +One command composes every evidence lane you have into one fail-closed, +content-addressed proof: contracts re-verified, suites re-run, before/after +movement measured, the stress suite cleared. The proof headlines its claim +scope, exactly what the evidence establishes: contracts alone re-measure stored +evidence (Captured Evidence), a suite or the stress suite establishes a Test +Suite ran, and a before/after run reaches Candidate Revision only when you bind +the candidate identity (`--candidate-config-hash`, `--provider`). CI gates on the exit code, and the receipt stays verifiable anywhere. ```bash diff --git a/docs/assets/hotato-cast.cast b/docs/assets/hotato-cast.cast deleted file mode 100644 index e04b431..0000000 --- a/docs/assets/hotato-cast.cast +++ /dev/null @@ -1,207 +0,0 @@ -{"version": 2, "width": 100, "height": 12, "timestamp": 1784237784, "env": {"SHELL": "/bin/bash", "TERM": "xterm-256color"}} -[0.70685, "o", "\u001b[1;32m$\u001b[0m p"] -[0.736142, "o", "y"] -[0.765269, "o", "t"] -[0.794733, "o", "h"] -[0.824266, "o", "o"] -[0.853648, "o", "n"] -[0.883111, "o", "3"] -[0.912902, "o", " "] -[0.942379, "o", "-"] -[0.971684, "o", "m"] -[1.001321, "o", " "] -[1.031185, "o", "h"] -[1.060415, "o", "o"] -[1.089508, "o", "t"] -[1.118924, "o", "a"] -[1.148044, "o", "t"] -[1.177044, "o", "o"] -[1.206138, "o", " "] -[1.235192, "o", "r"] -[1.26452, "o", "u"] -[1.293682, "o", "n"] -[1.323218, "o", " "] -[1.352376, "o", "-"] -[1.381968, "o", "-"] -[1.411312, "o", "s"] -[1.44053, "o", "t"] -[1.469911, "o", "e"] -[1.49917, "o", "r"] -[1.529012, "o", "e"] -[1.558153, "o", "o"] -[1.587233, "o", " "] -[1.616357, "o", "c"] -[1.645445, "o", "o"] -[1.674653, "o", "r"] -[1.703741, "o", "p"] -[1.733203, "o", "u"] -[1.762695, "o", "s"] -[1.791975, "o", "/"] -[1.821095, "o", "r"] -[1.850267, "o", "e"] -[1.879708, "o", "a"] -[1.908774, "o", "l"] -[1.938261, "o", "/"] -[1.967359, "o", "a"] -[1.996569, "o", "u"] -[2.025948, "o", "d"] -[2.055203, "o", "i"] -[2.084494, "o", "o"] -[2.113613, "o", "/"] -[2.14275, "o", "a"] -[2.172229, "o", "m"] -[2.201253, "o", "i"] -[2.230431, "o", "-"] -[2.259573, "o", "e"] -[2.288961, "o", "n"] -[2.318388, "o", "2"] -[2.347973, "o", "0"] -[2.377548, "o", "0"] -[2.406969, "o", "2"] -[2.436649, "o", "b"] -[2.466115, "o", "-"] -[2.495541, "o", "t"] -[2.524754, "o", "a"] -[2.553985, "o", "k"] -[2.583229, "o", "e"] -[2.612476, "o", "-"] -[2.641834, "o", "0"] -[2.671028, "o", "1"] -[2.700348, "o", "4"] -[2.729536, "o", "9"] -[2.759185, "o", "."] -[2.788335, "o", "e"] -[2.817522, "o", "x"] -[2.846747, "o", "a"] -[2.875855, "o", "m"] -[2.905165, "o", "p"] -[2.935307, "o", "l"] -[2.964781, "o", "e"] -[2.994063, "o", "."] -[3.023166, "o", "w"] -[3.052424, "o", "a"] -[3.081598, "o", "v"] -[3.110836, "o", " "] -[3.140188, "o", "\\"] -[3.169507, "o", "\r\n"] -[3.169549, "o", "\u001b[1;32m>\u001b[0m "] -[3.169579, "o", " "] -[3.198742, "o", " "] -[3.228337, "o", " "] -[3.257471, "o", " "] -[3.287013, "o", "-"] -[3.316091, "o", "-"] -[3.34511, "o", "o"] -[3.374163, "o", "n"] -[3.403405, "o", "s"] -[3.432548, "o", "e"] -[3.461985, "o", "t"] -[3.491277, "o", " "] -[3.520576, "o", "3"] -[3.549805, "o", "."] -[3.579079, "o", "0"] -[3.608422, "o", " "] -[3.638179, "o", "-"] -[3.667299, "o", "-"] -[3.696547, "o", "e"] -[3.726294, "o", "x"] -[3.75553, "o", "p"] -[3.784725, "o", "e"] -[3.81399, "o", "c"] -[3.843186, "o", "t"] -[3.87248, "o", " "] -[3.901651, "o", "y"] -[3.930838, "o", "i"] -[3.959975, "o", "e"] -[3.989107, "o", "l"] -[4.018632, "o", "d"] -[4.048204, "o", " "] -[4.077269, "o", "-"] -[4.106383, "o", "-"] -[4.135579, "o", "m"] -[4.16477, "o", "a"] -[4.193989, "o", "x"] -[4.223261, "o", "-"] -[4.252604, "o", "t"] -[4.281812, "o", "a"] -[4.31109, "o", "l"] -[4.340222, "o", "k"] -[4.36927, "o", "-"] -[4.398448, "o", "o"] -[4.427627, "o", "v"] -[4.4569, "o", "e"] -[4.486565, "o", "r"] -[4.51622, "o", " "] -[4.545645, "o", "1"] -[4.575034, "o", "."] -[4.60441, "o", "8"] -[4.633683, "o", "5"] -[4.663021, "o", " "] -[4.69217, "o", "-"] -[4.721544, "o", "-"] -[4.750724, "o", "m"] -[4.779953, "o", "a"] -[4.809258, "o", "x"] -[4.83882, "o", "-"] -[4.86802, "o", "t"] -[4.897307, "o", "i"] -[4.926487, "o", "m"] -[4.95601, "o", "e"] -[4.985165, "o", "-"] -[5.014291, "o", "t"] -[5.043341, "o", "o"] -[5.072777, "o", "-"] -[5.10233, "o", "y"] -[5.13161, "o", "i"] -[5.160889, "o", "e"] -[5.190082, "o", "l"] -[5.219309, "o", "d"] -[5.248931, "o", " "] -[5.278164, "o", "1"] -[5.307247, "o", "."] -[5.336609, "o", "8"] -[5.366183, "o", "5"] -[5.395317, "o", " "] -[5.424467, "o", "\\"] -[5.453586, "o", "\r\n"] -[5.453645, "o", "\u001b[1;32m>\u001b[0m "] -[5.453672, "o", " "] -[5.482926, "o", " "] -[5.512103, "o", " "] -[5.541345, "o", " "] -[5.570561, "o", "-"] -[5.599728, "o", "-"] -[5.628879, "o", "c"] -[5.658549, "o", "o"] -[5.687757, "o", "n"] -[5.717079, "o", "f"] -[5.746325, "o", "i"] -[5.775493, "o", "r"] -[5.804623, "o", "m"] -[5.833833, "o", "-"] -[5.86309, "o", "c"] -[5.892329, "o", "h"] -[5.921462, "o", "a"] -[5.950686, "o", "n"] -[5.980267, "o", "n"] -[6.009798, "o", "e"] -[6.039009, "o", "l"] -[6.068561, "o", "s"] -[6.09799, "o", " "] -[6.12759, "o", "-"] -[6.157036, "o", "-"] -[6.186152, "o", "f"] -[6.215373, "o", "o"] -[6.24456, "o", "r"] -[6.273729, "o", "m"] -[6.302862, "o", "a"] -[6.331941, "o", "t"] -[6.361084, "o", " "] -[6.390869, "o", "t"] -[6.420102, "o", "e"] -[6.449368, "o", "x"] -[6.47854, "o", "t"] -[6.507626, "o", "\r\n"] -[7.085756, "o", "hotato [single] stack=generic offline=True\r\n 1/1 events pass (failed=0)\r\n [PASS] ami-en2002b-take-0149.example.wav: did_yield=True seconds_to_yield=1.25s talk_over=0.58s\r\n exit_code=0\r\n"] -[7.518234, "o", "\u001b[1;32m$\u001b[0m "] -[9.719899, "o", "\r\n"] diff --git a/docs/assets/hotato-cast.gif b/docs/assets/hotato-cast.gif deleted file mode 100644 index 602535a..0000000 Binary files a/docs/assets/hotato-cast.gif and /dev/null differ diff --git a/llms-full.txt b/llms-full.txt index 13d9b20..e10409e 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -5315,7 +5315,7 @@ reports pass/fail. | **Re-scores** | the SAME `audio/event.wav` the contract was created from | a new recording of the SAME stimulus against your CURRENT agent | | **A pass proves** | the evidence, policy, and scorer are still intact and still agree with the human label | the CURRENT agent's behavior on that stimulus still matches the label | | **Leaves open** | whether the deployed agent's behavior has changed since | nothing extra -- this lane speaks to the live agent | -| **Runs** | every push, in the shipped `ci/github-action.yml` (`contract verify contracts/`) | only when you recapture by hand or on a schedule -- see [`docs/RECAPTURE.md`](RECAPTURE.md) | +| **Runs** | every push, in the shipped `ci/github_action.yml` (`contract verify contracts/`) | only when you recapture by hand or on a schedule -- see [`docs/RECAPTURE.md`](RECAPTURE.md) | A frozen-recording pass is necessary but not sufficient: the recording never changes, so it can only fail if someone edits the bundle's audio @@ -5436,7 +5436,7 @@ Exit codes are the CI contract: | `2` | usage error, empty directory, or corrupt `contract.json` | `--junit` writes one `` per contract; the shipped -`ci/github-action.yml` scaffold runs this on push, on PR, and weekly, +`ci/github_action.yml` scaffold runs this on push, on PR, and weekly, and publishes the JUnit file as an artifact. Every text and HTML render of `verify` also prints, verbatim: *"This @@ -5525,7 +5525,7 @@ matching the packed manifest. ## CI -The shipped `ci/github-action.yml` is the minimal wiring: +The shipped `ci/github_action.yml` is the minimal wiring: ```bash uvx hotato contract verify contracts/ --junit contracts-junit.xml \ @@ -9895,9 +9895,13 @@ deterministic: the same scenario and seed render the same bytes on every run. ## 5. Prove -One command composes every evidence lane you have into one fail-closed release -proof: contracts re-verified, suites re-run, before/after movement measured, -the stress suite cleared. The proof is a content-addressed receipt; CI gates on +One command composes every evidence lane you have into one fail-closed, +content-addressed proof: contracts re-verified, suites re-run, before/after +movement measured, the stress suite cleared. The proof headlines its claim +scope, exactly what the evidence establishes: contracts alone re-measure stored +evidence (Captured Evidence), a suite or the stress suite establishes a Test +Suite ran, and a before/after run reaches Candidate Revision only when you bind +the candidate identity (`--candidate-config-hash`, `--provider`). CI gates on the exit code, and the receipt stays verifiable anywhere. ```bash diff --git a/llms.txt b/llms.txt index 798a407..fb41c8f 100644 --- a/llms.txt +++ b/llms.txt @@ -9,7 +9,9 @@ > finds what text evals miss), pin (a labeled failure becomes a portable > content-addressed contract), test (simulate, drive, and the bundled stress > suite), and prove (hotato prove composes every evidence lane into one -> fail-closed release proof). The wedge leads the demo: give it a two-channel recording or a +> fail-closed, content-addressed proof headlining its claim scope: contracts +> alone are Captured Evidence, a suite is a Test Suite, and a bound before/after +> run is a Candidate Revision, never over-claiming). The wedge leads the demo: give it a two-channel recording or a > timestamped transcript and it measures turn timing and say-do evidence -- did > the agent yield when the caller took the floor, how fast, how many seconds > both talked at once, and did what the agent said match what the backend did. diff --git a/saydo/state.json b/saydo/state.json deleted file mode 100644 index 266d8b8..0000000 --- a/saydo/state.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "orders": [ - { - "order_id": "A-3090", - "refund_status": "none" - } - ] -} diff --git a/saydo/test-run.json b/saydo/test-run.json deleted file mode 100644 index c7380bf..0000000 --- a/saydo/test-run.json +++ /dev/null @@ -1,155 +0,0 @@ -{ - "kind": "hotato.test-run", - "version": 1, - "test_id": "demo-refund-claimed-not-issued", - "agent": "demo-agent", - "inconclusive_policy": "report", - "exit_code": 1, - "success": { - "required": [ - "all_deterministic_assertions_pass" - ], - "conditions": { - "all_deterministic_assertions_pass": false - }, - "passed": false, - "rubric_gated": false - }, - "assertions": { - "schema": "assert.v1", - "exit_code": 1, - "inconclusive_policy": "report", - "results": [ - { - "id": "agent-said-refund-sent", - "kind": "phrase", - "deterministic": true, - "dimension": "conversation", - "status": "PASS" - }, - { - "id": "outcome-refund-tool", - "kind": "tool_result", - "deterministic": true, - "dimension": "outcome", - "status": "FAIL", - "reason": "tool 'issue_refund' produced no result span in the trace", - "public_reason": "issue_refund produced no result satisfying the declared conditions." - }, - { - "id": "outcome-refund-state", - "kind": "state", - "deterministic": true, - "dimension": "outcome", - "status": "FAIL", - "reason": "'orders' record present but field(s) ['refund_status'] did not match the expected post-call state", - "public_reason": "orders post-call state did not satisfy the declared fields." - } - ], - "summary": { - "deterministic": { - "pass": 1, - "fail": 2, - "inconclusive": 0 - }, - "judge": { - "pass": 0, - "fail": 0 - }, - "note": "inconclusive_policy=report: 1 pass, 2 fail, 0 inconclusive across 3 deterministic assertion(s); 0 judge-scored assertions (a judge kind is a separate, quarantined capability, not built here)" - } - }, - "rubric": { - "schema": "rubric.v1", - "exit_code": 0, - "advisory": true, - "gated": false, - "results": [], - "summary": { - "pass": 0, - "fail": 0, - "inconclusive": 0, - "error": 0, - "note": "0 pass, 0 fail, 0 inconclusive, 0 error across 0 rubric result(s); model-judged (advisory), deterministic:false, never merged into the deterministic counts and never a blended or overall number. ADVISORY: no verdict gates CI." - } - }, - "dimensions": { - "outcome": { - "pass": 0, - "fail": 2, - "inconclusive": 0, - "ids": [ - "outcome-refund-tool", - "outcome-refund-state" - ] - }, - "policy": { - "pass": 0, - "fail": 0, - "inconclusive": 0, - "ids": [] - }, - "conversation": { - "pass": 1, - "fail": 0, - "inconclusive": 0, - "ids": [ - "agent-said-refund-sent" - ] - }, - "speech": { - "pass": 0, - "fail": 0, - "inconclusive": 0, - "ids": [] - }, - "reliability": { - "pass": 0, - "fail": 0, - "inconclusive": 0, - "ids": [] - } - }, - "reliability": { - "aggregate": { - "pass_at_1": 0.0, - "pass_at_k": 0.0, - "pass_caret_k": 0.0, - "n": 1, - "k": 1, - "passes": 0, - "ci": { - "low": 0.0, - "high": 0.793457, - "method": "wilson", - "z": 1.96 - }, - "note": "reliability over 1 repeated run(s) of the deterministic lane on the same supplied fixture recording; pass^k == pass@1 because the deterministic replay is byte-identical (zero run-to-run variance)" - }, - "origin": "fixture", - "runs": 1, - "basis": "agent_deterministic_replay", - "note": "reliability over 1 repeated run(s) of the deterministic lane on the same supplied fixture recording; pass^k == pass@1 because the deterministic replay is byte-identical (zero run-to-run variance)", - "per_run": [ - { - "run": 1, - "exit_code": 1, - "passed": false - } - ] - }, - "repetitions": { - "runs": 1, - "per_run": [ - { - "run": 1, - "exit_code": 1, - "summary": { - "pass": 1, - "fail": 2, - "inconclusive": 0 - } - } - ] - } -} diff --git a/saydo/test.json b/saydo/test.json deleted file mode 100644 index d3ceb67..0000000 --- a/saydo/test.json +++ /dev/null @@ -1,49 +0,0 @@ -{ - "agent": "demo-agent", - "assertions": { - "deterministic": [ - { - "dimension": "conversation", - "id": "agent-said-refund-sent", - "kind": "phrase", - "regex": "refund .*sent", - "role": "agent" - }, - { - "dimension": "outcome", - "id": "outcome-refund-tool", - "kind": "tool_result", - "name": "issue_refund", - "result_subset": { - "status": "refunded" - } - }, - { - "dimension": "outcome", - "expect": { - "refund_status": "refunded" - }, - "filters": { - "order_id": "A-3090" - }, - "id": "outcome-refund-state", - "kind": "state", - "resource": "orders" - } - ] - }, - "id": "demo-refund-claimed-not-issued", - "inconclusive_policy": "report", - "kind": "hotato.conversation-test", - "repetitions": 1, - "success": { - "report_dimensions": [ - "outcome", - "conversation" - ], - "required": [ - "all_deterministic_assertions_pass" - ] - }, - "version": 1 -} diff --git a/saydo/trace.jsonl b/saydo/trace.jsonl deleted file mode 100644 index d21e591..0000000 --- a/saydo/trace.jsonl +++ /dev/null @@ -1,4 +0,0 @@ -{"_meta": true, "call_id": null, "created_by": "hotato demo say-do bundle (scripted conversation)", "deployment": {"agent_id": null, "config_hash": null, "git_sha": null, "stack": "demo"}, "schema": "hotato.voice_trace.v1", "source": {"format": "scripted-demo", "span_count": 3}} -{"end_sec": 2.68, "start_sec": 0.6, "type": "caller_audio_active"} -{"arguments": {"order_id": "A-3090"}, "end_sec": 6.2, "latency_ms": 300, "name": "lookup_order", "result": {"found": true}, "start_sec": 5.9, "type": "tool_call"} -{"end_sec": 8.72, "start_sec": 6.16, "type": "caller_audio_active"} diff --git a/saydo/transcript.json b/saydo/transcript.json deleted file mode 100644 index c9a1cf3..0000000 --- a/saydo/transcript.json +++ /dev/null @@ -1,28 +0,0 @@ -{ - "segments": [ - { - "end": 2.68, - "role": "caller", - "start": 0.6, - "text": "order A-3090 never arrived" - }, - { - "end": 5.66, - "role": "agent", - "start": 3.18, - "text": "sorry about that, let me pull up order A-3090" - }, - { - "end": 8.72, - "role": "caller", - "start": 6.16, - "text": "i want a refund for order A-3090" - }, - { - "end": 11.86, - "role": "agent", - "start": 9.22, - "text": "done, your refund for order A-3090 has been sent" - } - ] -}