From f957af072bd42d563a08efc97884726bf184ec00 Mon Sep 17 00:00:00 2001 From: tgasser-nv <200644301+tgasser-nv@users.noreply.github.com> Date: Fri, 2 Oct 2026 09:15:39 -0500 Subject: [PATCH 1/9] Check in tests (expected to fail) --- tests/guardrails/test_iorails_check.py | 74 ++++++++++++ .../test_transform_rail_pipeline.py | 110 +++++++++++++++++- 2 files changed, 181 insertions(+), 3 deletions(-) diff --git a/tests/guardrails/test_iorails_check.py b/tests/guardrails/test_iorails_check.py index 2f753aadd5..7fa0f932c1 100644 --- a/tests/guardrails/test_iorails_check.py +++ b/tests/guardrails/test_iorails_check.py @@ -836,6 +836,80 @@ async def test_a_block_behind_a_rewrite_is_reported_as_blocked(self, iorails): assert result.rail == "content safety check output" +@pytest.mark.asyncio +class TestCheckContentCaptureRecordsMaskedMessages: + """Capture records the checked messages as the rails masked them, so a span cannot carry what a mask removed.""" + + async def test_an_input_mask_is_captured_masked(self, iorails): + """The captured user message is the one the input rails masked.""" + iorails._content_capture_enabled = True + iorails.rails_manager.is_input_safe = AsyncMock(return_value=user_message_rewrite(MASKED_USER_TEXT)) + + with patch("nemoguardrails.guardrails.iorails.set_request_content") as capture: + await iorails.check_async([{"role": "user", "content": USER_TEXT}]) + + assert capture.call_args.args[1] == [{"role": "user", "content": MASKED_USER_TEXT}] + + async def test_an_output_mask_is_captured_masked(self, iorails): + """The captured assistant message is the one the output rails masked.""" + iorails._content_capture_enabled = True + iorails.rails_manager.is_output_safe = AsyncMock(return_value=bot_message_rewrite(MASKED_BOT_TEXT)) + + with patch("nemoguardrails.guardrails.iorails.set_request_content") as capture: + await iorails.check_async(CONVERSATION, rail_types=[RailType.OUTPUT]) + + assert capture.call_args.args[1] == [ + {"role": "user", "content": USER_TEXT}, + {"role": "assistant", "content": MASKED_BOT_TEXT}, + ] + + async def test_both_masks_are_captured_masked(self, iorails): + """With both directions masked, neither raw text reaches the span.""" + iorails._content_capture_enabled = True + iorails.rails_manager.is_input_safe = AsyncMock(return_value=user_message_rewrite(MASKED_USER_TEXT)) + iorails.rails_manager.is_output_safe = AsyncMock(return_value=bot_message_rewrite(MASKED_BOT_TEXT)) + + with patch("nemoguardrails.guardrails.iorails.set_request_content") as capture: + await iorails.check_async(CONVERSATION) + + assert capture.call_args.args[1] == [ + {"role": "user", "content": MASKED_USER_TEXT}, + {"role": "assistant", "content": MASKED_BOT_TEXT}, + ] + + async def test_an_input_mask_is_captured_masked_when_the_output_blocks(self, iorails): + """A block after an input mask still captures the masked user message.""" + iorails._content_capture_enabled = True + iorails.rails_manager.is_input_safe = AsyncMock(return_value=user_message_rewrite(MASKED_USER_TEXT)) + iorails.rails_manager.is_output_safe = AsyncMock(return_value=_unsafe("content safety check output")) + + with patch("nemoguardrails.guardrails.iorails.set_request_content") as capture: + await iorails.check_async(CONVERSATION) + + assert capture.call_args.args[1][0] == {"role": "user", "content": MASKED_USER_TEXT} + + async def test_an_unmasked_check_is_captured_as_it_arrived(self, iorails): + """With no rewrite, the span records the messages the caller sent.""" + iorails._content_capture_enabled = True + _mock_rails(iorails) + + with patch("nemoguardrails.guardrails.iorails.set_request_content") as capture: + await iorails.check_async(CONVERSATION) + + assert capture.call_args.args[1] == CONVERSATION + + async def test_an_output_mask_leaves_the_callers_messages_unchanged(self, iorails): + """Masking for the span copies the conversation rather than editing the caller's list.""" + iorails._content_capture_enabled = True + iorails.rails_manager.is_output_safe = AsyncMock(return_value=bot_message_rewrite(MASKED_BOT_TEXT)) + messages = [dict(message) for message in CONVERSATION] + + with patch("nemoguardrails.guardrails.iorails.set_request_content"): + await iorails.check_async(messages, rail_types=[RailType.OUTPUT]) + + assert messages == CONVERSATION + + class TestUnsatisfiableRailTypes: """Requesting a rail type with no configured flows raises RailTypeNotConfiguredError.""" diff --git a/tests/guardrails/test_transform_rail_pipeline.py b/tests/guardrails/test_transform_rail_pipeline.py index d2473dc9c2..5c512410b8 100644 --- a/tests/guardrails/test_transform_rail_pipeline.py +++ b/tests/guardrails/test_transform_rail_pipeline.py @@ -15,9 +15,9 @@ """A masking rail among judging rails, driven end to end through IORails. -Covers both directions, both concurrency settings and streaming. Nothing is stubbed at the rail -boundary -- the shipped actions run against canned model and HTTP replies -- so a rail that -stopped reading its conversation variable fails here rather than in a config. +Covers both directions, both concurrency settings, streaming and rails-only checks. Nothing is +stubbed at the rail boundary -- the shipped actions run against canned model and HTTP replies -- +so a rail that stopped reading its conversation variable fails here rather than in a config. """ import copy @@ -27,10 +27,16 @@ import httpx import pytest import pytest_asyncio +from opentelemetry.sdk.trace import ReadableSpan, TracerProvider +from opentelemetry.sdk.trace.export import SimpleSpanProcessor +from opentelemetry.sdk.trace.export.in_memory_span_exporter import InMemorySpanExporter +from nemoguardrails.guardrails import telemetry from nemoguardrails.guardrails.guardrails_types import RailDirection from nemoguardrails.guardrails.iorails import REFUSAL_MESSAGE, IORails from nemoguardrails.rails.llm.config import RailsConfig +from nemoguardrails.rails.llm.options import RailType +from nemoguardrails.tracing.constants import GuardrailsAttributes from nemoguardrails.types import LLMResponse, LLMResponseChunk from tests.guardrails.async_helpers import JAILBREAK_NIM_URL, started_iorails from tests.guardrails.test_data import NEMOGUARDS_CONFIG @@ -566,3 +572,101 @@ async def test_the_output_rail_reads_the_masked_user_message(self, call_log, htt assert MASKED_INPUT in call_log.text_seen_by(CONTENT_SAFETY_RAIL) assert PERSON not in call_log.text_seen_by(CONTENT_SAFETY_RAIL) + + +CHECKED_CONVERSATION = [ + {"role": "user", "content": USER_INPUT}, + {"role": "assistant", "content": MAIN_OUTPUT_WITH_PII}, +] +# One model answers content safety in both directions, so its verdict carries both keys. +CHECK_ALL_ALLOW = { + CONTENT_SAFETY_RAIL: SAFE_OUTPUT_VERDICT, + TOPIC_CONTROL_RAIL: ON_TOPIC_VERDICT, + SELF_CHECK_RAIL: SELF_CHECK_ALLOWS, +} + + +def _check_capture_config() -> dict: + """Both pipelines in one config, traced with content capture on so a check's request span records content.""" + config = _output_pipeline_config() + config["rails"]["input"] = {"flows": list(INPUT_FLOWS)} + config["rails"]["config"]["gliner"]["input"] = {"entities": ["person"]} + config["tracing"] = {"enabled": True, "enable_content_capture": True} + return config + + +@pytest.fixture +def span_exporter() -> InMemorySpanExporter: + return InMemorySpanExporter() + + +@pytest_asyncio.fixture +async def check_capture_iorails(span_exporter): + """IORails masking in both directions, its spans exported to ``span_exporter``.""" + provider = TracerProvider() + provider.add_span_processor(SimpleSpanProcessor(span_exporter)) + # IORails takes its tracer when it is built, so the test tracer has to be in place first. + with patch.object(telemetry, "_tracer", provider.get_tracer("test")): + async with started_iorails(_check_capture_config()) as engine: + yield engine + + +def _request_span(exporter: InMemorySpanExporter) -> ReadableSpan: + """The one ``guardrails.request`` span the check produced.""" + request_spans = [span for span in exporter.get_finished_spans() if span.name == "guardrails.request"] + assert len(request_spans) == 1 + return request_spans[0] + + +def _captured_input(exporter: InMemorySpanExporter) -> list[dict]: + """The messages the check's request span recorded.""" + return json.loads(_request_span(exporter).attributes[GuardrailsAttributes.REQUEST_INPUT]) + + +def _captured_output(exporter: InMemorySpanExporter) -> str: + """The text the check's request span recorded as returned.""" + return _request_span(exporter).attributes[GuardrailsAttributes.REQUEST_OUTPUT] + + +@pytest.mark.asyncio +class TestACheckRecordsTheMaskedConversation: + """With content capture on, a check's request span records the conversation as the masking rails left it.""" + + async def test_the_input_mask_reaches_the_request_span( + self, check_capture_iorails, call_log, httpx_mock, span_exporter + ): + """The span records the user's message as the input mask left it.""" + _wire(check_capture_iorails, call_log, httpx_mock, CHECK_ALL_ALLOW) + + await check_capture_iorails.check_async([{"role": "user", "content": USER_INPUT}]) + + assert _captured_input(span_exporter) == [{"role": "user", "content": MASKED_INPUT}] + assert _captured_output(span_exporter) == MASKED_INPUT + + async def test_the_output_mask_reaches_the_request_span( + self, check_capture_iorails, call_log, httpx_mock, span_exporter + ): + """A check's conversation carries the response it checked, so the output mask reaches the span too.""" + _wire(check_capture_iorails, call_log, httpx_mock, CHECK_ALL_ALLOW) + + await check_capture_iorails.check_async(CHECKED_CONVERSATION, rail_types=[RailType.OUTPUT]) + + assert _captured_input(span_exporter) == [ + {"role": "user", "content": USER_INPUT}, + {"role": "assistant", "content": MASKED_MAIN_OUTPUT}, + ] + assert _captured_output(span_exporter) == MASKED_MAIN_OUTPUT + + # GLiNER masks once per direction, so its one callback has to answer twice. + @pytest.mark.httpx_mock(can_send_already_matched_responses=True) + async def test_both_masks_reach_the_request_span(self, check_capture_iorails, call_log, httpx_mock, span_exporter): + """Masked in both directions, the name reaches the span from neither turn.""" + _wire(check_capture_iorails, call_log, httpx_mock, CHECK_ALL_ALLOW) + + await check_capture_iorails.check_async(CHECKED_CONVERSATION) + + assert _captured_input(span_exporter) == [ + {"role": "user", "content": MASKED_INPUT}, + {"role": "assistant", "content": MASKED_MAIN_OUTPUT}, + ] + assert _captured_output(span_exporter) == MASKED_MAIN_OUTPUT From d6234038c155b61c6381bc0626ebc4f633b9b557 Mon Sep 17 00:00:00 2001 From: tgasser-nv <200644301+tgasser-nv@users.noreply.github.com> Date: Fri, 2 Oct 2026 09:48:37 -0500 Subject: [PATCH 2/9] Add code to mask last assistant message --- .../observability/tracing/content-capture.mdx | 6 +++- docs/observability/tracing/span-reference.mdx | 2 +- .../using-python-apis/check-messages.mdx | 1 + nemoguardrails/guardrails/iorails.py | 23 ++++++++++++-- nemoguardrails/guardrails/telemetry.py | 3 +- tests/guardrails/test_iorails_check.py | 30 +++++++++++++++++++ 6 files changed, 60 insertions(+), 5 deletions(-) diff --git a/docs/observability/tracing/content-capture.mdx b/docs/observability/tracing/content-capture.mdx index 92afe72541..c9c327e726 100644 --- a/docs/observability/tracing/content-capture.mdx +++ b/docs/observability/tracing/content-capture.mdx @@ -31,7 +31,7 @@ When content capture is enabled, IORails records content on three types of spans | Span (Kind) | Attribute or Event | When Recorded | |-------------|--------------------|---------------| -| `guardrails.request` (SERVER) | `guardrails.request.input`: JSON-encoded list of the caller's input messages | Always, while capturing | +| `guardrails.request` (SERVER) | `guardrails.request.input`: JSON-encoded list of the input messages, after any input-rail rewrite (and, on a check, any output-rail rewrite) | Always, while capturing | | `guardrails.request` (SERVER) | `guardrails.request.output`: the text actually returned to the caller | When output is produced (a blocked request records the refusal message; an empty stream records nothing) | | `chat ` (CLIENT) | LLM input and output, in the [selected format](#output-format) | Once per LLM call: the main generation call and every rail-action LLM call. On the streaming path, recorded when the stream ends; see [Streaming](#streaming) | | `guardrails.rail` (INTERNAL) | `guardrails.rail.input`: JSON-encoded `{"messages": ..., "bot_response": ...}` snapshot | Every rail execution, while capturing | @@ -44,6 +44,7 @@ Sampling parameters (`gen_ai.request.temperature`, `max_tokens`, and so on) and A rails-only [`check_async()`](/run-guardrailed-inference/using-python-apis/check-messages) uses the same capture behavior. The checked messages are recorded in `guardrails.request.input`, and the returned content is recorded in `guardrails.request.output`. +When an input or output rail rewrites the text it checked, `guardrails.request.input` carries the rewritten messages. Every rail that runs records its own `guardrails.rail` attributes. A check makes no main-model call, so it produces no CLIENT span for the protected model. `IORails` captures LLM calls from rail actions as usual. @@ -229,6 +230,9 @@ In both cases, if no content was produced, IORails omits the output rather than - Content capture is **off by default**. Prompts and responses are never written to spans unless you explicitly enable it. - Captured content can include PII or other sensitive data. Treat your tracing backend as a system that stores user content, and apply the same access controls and retention limits you would apply to any store of conversation data. +- Masking keeps text out of `guardrails.request.input`, but not out of the `guardrails.rail` spans, which record the input each rail received. + The masking rail's own span carries the text before the mask. + On a `check_async()` call, whose conversation includes the checked response, every rail that runs records that response as it was sent. - Use `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=false` to force capture off across every service from the environment, regardless of what individual configs request. This is useful as a deployment-wide guardrail for regulated environments. - The [OpenTelemetry GenAI semantic conventions](https://opentelemetry.io/docs/specs/semconv/gen-ai/) recommend against capturing message content by default because of these privacy risks. diff --git a/docs/observability/tracing/span-reference.mdx b/docs/observability/tracing/span-reference.mdx index bd347502f2..6ec5423fd0 100644 --- a/docs/observability/tracing/span-reference.mdx +++ b/docs/observability/tracing/span-reference.mdx @@ -240,7 +240,7 @@ Every span sets ERROR status and records the exception when one propagates throu | --------------------------- | ------ | ------------------ | --------------------------------------------------------------------------------------------------------------------- | | `gen_ai.operation.name` | string | Always | The value `guardrails`, marking the request boundary. | | `request.id` | string | Always | Request identifier derived from the OpenTelemetry trace ID. | -| `guardrails.request.input` | string | Content capture on | JSON-encoded input messages received from the caller. | +| `guardrails.request.input` | string | Content capture on | JSON-encoded input messages, after any input-rail rewrite (and, on a check, any output-rail rewrite). | | `guardrails.request.output` | string | Content capture on | The text actually returned to the caller. On a blocked request this is the refusal message, not the raw model output. | When speculative generation is active, the request span also carries these attributes: diff --git a/docs/run-rails/using-python-apis/check-messages.mdx b/docs/run-rails/using-python-apis/check-messages.mdx index 1abb1d44df..bdfd43509c 100644 --- a/docs/run-rails/using-python-apis/check-messages.mdx +++ b/docs/run-rails/using-python-apis/check-messages.mdx @@ -427,6 +427,7 @@ When your application requires that a rail actually ran, assert that the message A check rejected by a full admission queue increments `guardrails.nonstream.rejections`. For more information, refer to [Metric Reference](/observability/metrics/reference). - Content capture records the checked messages in `guardrails.request.input` and the returned content in `guardrails.request.output` on the check's request span. + When an input or output rail rewrites the text it checked, for example to mask personal data, the span records the rewritten messages rather than the ones the caller sent. For more information, refer to [Capturing Prompt and Response Content](/observability/tracing/content-capture). The synchronous `check()` builds a short-lived engine with tracing and metrics disabled, so it emits no spans and no metrics. Use `check_async()` on instrumented paths. diff --git a/nemoguardrails/guardrails/iorails.py b/nemoguardrails/guardrails/iorails.py index df625f71cd..1f54a133ef 100644 --- a/nemoguardrails/guardrails/iorails.py +++ b/nemoguardrails/guardrails/iorails.py @@ -592,6 +592,16 @@ def _reports_assistant_content(rails_to_run: list[str]) -> bool: return "tool_call" in rails_to_run and "input" not in rails_to_run +def _rewrite_last_assistant_message(messages: LLMMessages, text: str) -> LLMMessages: + """Return a copy of *messages* with the last assistant message's content rewritten to *text*.""" + for index in range(len(messages) - 1, -1, -1): + if messages[index].get("role") == "assistant": + rewritten = list(messages) + rewritten[index] = {**messages[index], "content": text} + return rewritten + raise ValueError("no assistant message to rewrite") + + def _blocked_message(result: RailResult) -> str: """The text a blocked turn returns, which says whether the rail broke or fired.""" if result.failed: @@ -1466,16 +1476,19 @@ async def _run_check( ) -> RailsResult: """Queue-worker entry for ``check_async``: wrap the rails in a request span.""" tracer = self._tracer if self._tracing_enabled else None + conversation = _TurnConversation(messages=messages) with traced_request(tracer) as (request_span, req_id): t0 = time.monotonic() try: - result = await self._do_check(messages, rail_types, req_id, tools=tools) + result = await self._do_check(messages, rail_types, req_id, tools=tools, conversation=conversation) except Exception: elapsed_ms = (time.monotonic() - t0) * 1000 log.error("[%s] check failed time=%.1fms", req_id, elapsed_ms, exc_info=True) raise + # The messages as the rails masked them, as _run_generate captures them. A check's + # messages carry the checked response too, so an output mask must reach them as well. if self._content_capture_enabled: - set_request_content(request_span, messages, result.content) + set_request_content(request_span, conversation.messages, result.content) elapsed_ms = (time.monotonic() - t0) * 1000 log.info( "[%s] check completed time=%.1fms status=%s", @@ -1492,6 +1505,7 @@ async def _do_check( req_id: str, *, tools: Optional[list[dict]] = None, + conversation: Optional["_TurnConversation"] = None, ) -> RailsResult: """Core check pipeline: run the requested input, output and tool rails on messages.""" log.info("[%s] check called", req_id) @@ -1538,6 +1552,8 @@ async def _do_check( if rewritten is not None: log.info("[%s] Input rails rewrote the user message", req_id) messages = rewrite_user_message(messages, rewritten) + if conversation is not None: + conversation.messages = messages if not reports_output: pass_content = rewritten else: @@ -1564,6 +1580,9 @@ async def _do_check( if rewritten is not None: log.info("[%s] Output rails rewrote the response", req_id) pass_content = rewritten + messages = _rewrite_last_assistant_message(messages, rewritten) + if conversation is not None: + conversation.messages = messages else: log.info("[%s] Output rails requested but no assistant content to check; skipping", req_id) diff --git a/nemoguardrails/guardrails/telemetry.py b/nemoguardrails/guardrails/telemetry.py index 9a9acd00d0..f3e6d8acc1 100644 --- a/nemoguardrails/guardrails/telemetry.py +++ b/nemoguardrails/guardrails/telemetry.py @@ -712,7 +712,8 @@ def set_request_content( different values and confuse backends correlating the two. ``guardrails.request.input`` is always a JSON-encoded list of role/content - message objects matching the caller's input. ``guardrails.request.output`` + message objects, as the rails left them: after any rewrite, such as a mask, + rather than as the caller sent them. ``guardrails.request.output`` is the plain string that IORails returned (REFUSAL_MESSAGE on block paths, the model's response text on the success path). ``output_text=None`` suppresses the output attribute entirely — used by the streaming path when diff --git a/tests/guardrails/test_iorails_check.py b/tests/guardrails/test_iorails_check.py index 7fa0f932c1..7146fc4875 100644 --- a/tests/guardrails/test_iorails_check.py +++ b/tests/guardrails/test_iorails_check.py @@ -35,6 +35,7 @@ IORails, _determine_rails_from_messages, _get_last_content_by_role, + _rewrite_last_assistant_message, ) from nemoguardrails.guardrails.rail_guard import rail_error_outcome from nemoguardrails.rails.llm.config import RailsConfig @@ -766,6 +767,35 @@ def test_get_last_content_by_role_none_content_returns_empty(self): """content=None on the matched message is normalized to ''.""" assert _get_last_content_by_role([{"role": "user", "content": None}], "user") == "" + def test_rewrite_last_assistant_message_rewrites_only_the_last(self): + """Only the last assistant message is rewritten, the one output rails checked.""" + messages = [ + {"role": "assistant", "content": "first"}, + {"role": "user", "content": "question"}, + {"role": "assistant", "content": "second"}, + ] + + rewritten = _rewrite_last_assistant_message(messages, "masked") + + assert rewritten == [ + {"role": "assistant", "content": "first"}, + {"role": "user", "content": "question"}, + {"role": "assistant", "content": "masked"}, + ] + + def test_rewrite_last_assistant_message_leaves_the_input_unchanged(self): + """The rewrite copies the list and the message rather than editing the caller's.""" + messages = [{"role": "user", "content": "question"}, {"role": "assistant", "content": "answer"}] + + _rewrite_last_assistant_message(messages, "masked") + + assert messages == [{"role": "user", "content": "question"}, {"role": "assistant", "content": "answer"}] + + def test_rewrite_last_assistant_message_raises_without_an_assistant_message(self): + """With no assistant message there is nothing a rewrite could apply to, so it fails loudly.""" + with pytest.raises(ValueError, match="no assistant message"): + _rewrite_last_assistant_message([{"role": "user", "content": "question"}], "masked") + USER_TEXT = "my ssn is 123-45-6789" MASKED_USER_TEXT = "my ssn is " From 81cc0cfc200ffcd2c90bf3638592d47e663caf6e Mon Sep 17 00:00:00 2001 From: tgasser-nv <200644301+tgasser-nv@users.noreply.github.com> Date: Fri, 2 Oct 2026 11:07:17 -0500 Subject: [PATCH 3/9] Check in tests (expected to fail) --- tests/guardrails/test_iorails_check.py | 2 +- .../test_transform_rail_pipeline.py | 123 ++++++++++++++---- 2 files changed, 101 insertions(+), 24 deletions(-) diff --git a/tests/guardrails/test_iorails_check.py b/tests/guardrails/test_iorails_check.py index 7146fc4875..729915d63d 100644 --- a/tests/guardrails/test_iorails_check.py +++ b/tests/guardrails/test_iorails_check.py @@ -868,7 +868,7 @@ async def test_a_block_behind_a_rewrite_is_reported_as_blocked(self, iorails): @pytest.mark.asyncio class TestCheckContentCaptureRecordsMaskedMessages: - """Capture records the checked messages as the rails masked them, so a span cannot carry what a mask removed.""" + """Capture records the checked messages as the rails masked them, so the request span cannot carry what a mask removed.""" async def test_an_input_mask_is_captured_masked(self, iorails): """The captured user message is the one the input rails masked.""" diff --git a/tests/guardrails/test_transform_rail_pipeline.py b/tests/guardrails/test_transform_rail_pipeline.py index 5c512410b8..8b97ed0cd0 100644 --- a/tests/guardrails/test_transform_rail_pipeline.py +++ b/tests/guardrails/test_transform_rail_pipeline.py @@ -22,6 +22,9 @@ import copy import json +import os +from contextlib import asynccontextmanager +from typing import AsyncIterator from unittest.mock import AsyncMock, patch import httpx @@ -36,7 +39,7 @@ from nemoguardrails.guardrails.iorails import REFUSAL_MESSAGE, IORails from nemoguardrails.rails.llm.config import RailsConfig from nemoguardrails.rails.llm.options import RailType -from nemoguardrails.tracing.constants import GuardrailsAttributes +from nemoguardrails.tracing.constants import GuardrailsAttributes, OtelContentCapture from nemoguardrails.types import LLMResponse, LLMResponseChunk from tests.guardrails.async_helpers import JAILBREAK_NIM_URL, started_iorails from tests.guardrails.test_data import NEMOGUARDS_CONFIG @@ -586,45 +589,56 @@ async def test_the_output_rail_reads_the_masked_user_message(self, call_log, htt } -def _check_capture_config() -> dict: - """Both pipelines in one config, traced with content capture on so a check's request span records content.""" +CAPTURE_CONTENT = {"enabled": True, "enable_content_capture": True} + + +def _capturing_config() -> dict: + """Both pipelines in one config, traced with content capture on so the request span records content.""" config = _output_pipeline_config() config["rails"]["input"] = {"flows": list(INPUT_FLOWS)} config["rails"]["config"]["gliner"]["input"] = {"entities": ["person"]} - config["tracing"] = {"enabled": True, "enable_content_capture": True} + config["tracing"] = dict(CAPTURE_CONTENT) return config +@asynccontextmanager +async def _capturing(config: dict, exporter: InMemorySpanExporter) -> AsyncIterator[IORails]: + """IORails started with its spans exported to *exporter*, capturing content as *config* alone decides.""" + provider = TracerProvider() + provider.add_span_processor(SimpleSpanProcessor(exporter)) + # IORails takes its tracer and resolves capture when it is built, and the env var would override the config. + with patch.dict("os.environ"), patch.object(telemetry, "_tracer", provider.get_tracer("test")): + os.environ.pop(OtelContentCapture.CAPTURE_CONTENT_ENV, None) + async with started_iorails(config) as engine: + yield engine + + @pytest.fixture def span_exporter() -> InMemorySpanExporter: return InMemorySpanExporter() @pytest_asyncio.fixture -async def check_capture_iorails(span_exporter): +async def capturing_iorails(span_exporter): """IORails masking in both directions, its spans exported to ``span_exporter``.""" - provider = TracerProvider() - provider.add_span_processor(SimpleSpanProcessor(span_exporter)) - # IORails takes its tracer when it is built, so the test tracer has to be in place first. - with patch.object(telemetry, "_tracer", provider.get_tracer("test")): - async with started_iorails(_check_capture_config()) as engine: - yield engine + async with _capturing(_capturing_config(), span_exporter) as engine: + yield engine def _request_span(exporter: InMemorySpanExporter) -> ReadableSpan: - """The one ``guardrails.request`` span the check produced.""" + """The one ``guardrails.request`` span the request produced.""" request_spans = [span for span in exporter.get_finished_spans() if span.name == "guardrails.request"] assert len(request_spans) == 1 return request_spans[0] def _captured_input(exporter: InMemorySpanExporter) -> list[dict]: - """The messages the check's request span recorded.""" + """The messages the request span recorded.""" return json.loads(_request_span(exporter).attributes[GuardrailsAttributes.REQUEST_INPUT]) def _captured_output(exporter: InMemorySpanExporter) -> str: - """The text the check's request span recorded as returned.""" + """The text the request span recorded as returned.""" return _request_span(exporter).attributes[GuardrailsAttributes.REQUEST_OUTPUT] @@ -633,23 +647,23 @@ class TestACheckRecordsTheMaskedConversation: """With content capture on, a check's request span records the conversation as the masking rails left it.""" async def test_the_input_mask_reaches_the_request_span( - self, check_capture_iorails, call_log, httpx_mock, span_exporter + self, capturing_iorails, call_log, httpx_mock, span_exporter ): """The span records the user's message as the input mask left it.""" - _wire(check_capture_iorails, call_log, httpx_mock, CHECK_ALL_ALLOW) + _wire(capturing_iorails, call_log, httpx_mock, CHECK_ALL_ALLOW) - await check_capture_iorails.check_async([{"role": "user", "content": USER_INPUT}]) + await capturing_iorails.check_async([{"role": "user", "content": USER_INPUT}]) assert _captured_input(span_exporter) == [{"role": "user", "content": MASKED_INPUT}] assert _captured_output(span_exporter) == MASKED_INPUT async def test_the_output_mask_reaches_the_request_span( - self, check_capture_iorails, call_log, httpx_mock, span_exporter + self, capturing_iorails, call_log, httpx_mock, span_exporter ): """A check's conversation carries the response it checked, so the output mask reaches the span too.""" - _wire(check_capture_iorails, call_log, httpx_mock, CHECK_ALL_ALLOW) + _wire(capturing_iorails, call_log, httpx_mock, CHECK_ALL_ALLOW) - await check_capture_iorails.check_async(CHECKED_CONVERSATION, rail_types=[RailType.OUTPUT]) + await capturing_iorails.check_async(CHECKED_CONVERSATION, rail_types=[RailType.OUTPUT]) assert _captured_input(span_exporter) == [ {"role": "user", "content": USER_INPUT}, @@ -659,14 +673,77 @@ async def test_the_output_mask_reaches_the_request_span( # GLiNER masks once per direction, so its one callback has to answer twice. @pytest.mark.httpx_mock(can_send_already_matched_responses=True) - async def test_both_masks_reach_the_request_span(self, check_capture_iorails, call_log, httpx_mock, span_exporter): + async def test_both_masks_reach_the_request_span(self, capturing_iorails, call_log, httpx_mock, span_exporter): """Masked in both directions, the name reaches the span from neither turn.""" - _wire(check_capture_iorails, call_log, httpx_mock, CHECK_ALL_ALLOW) + _wire(capturing_iorails, call_log, httpx_mock, CHECK_ALL_ALLOW) - await check_capture_iorails.check_async(CHECKED_CONVERSATION) + await capturing_iorails.check_async(CHECKED_CONVERSATION) assert _captured_input(span_exporter) == [ {"role": "user", "content": MASKED_INPUT}, {"role": "assistant", "content": MASKED_MAIN_OUTPUT}, ] assert _captured_output(span_exporter) == MASKED_MAIN_OUTPUT + + +CONTENT_SAFETY_INPUT_FLOW = "content safety check input $model=content_safety" +GLINER_INPUT_FLOW = "gliner mask pii on input" + + +def _capturing_streaming_config() -> dict: + """An input mask ahead of an input judge on the streaming path, traced with content capture on.""" + config = _streaming_config([CONTENT_SAFETY_OUTPUT_FLOW]) + config["rails"]["input"] = {"flows": [CONTENT_SAFETY_INPUT_FLOW, GLINER_INPUT_FLOW]} + config["rails"]["config"]["gliner"]["input"] = {"entities": ["person"]} + config["tracing"] = dict(CAPTURE_CONTENT) + return config + + +@pytest.mark.asyncio +class TestAMaskSurvivesABlockOnTheRequestSpan: + """A rail behind the mask refuses the request, and the request span still records the masked text.""" + + async def test_a_check_blocked_on_input_records_the_masked_user_message( + self, capturing_iorails, call_log, httpx_mock, span_exporter + ): + """The block decides the verdict, but the user's message it records is the masked one.""" + _wire(capturing_iorails, call_log, httpx_mock, CONTENT_SAFETY_BLOCKS) + + await capturing_iorails.check_async([{"role": "user", "content": USER_INPUT}]) + + assert _captured_input(span_exporter) == [{"role": "user", "content": MASKED_INPUT}] + assert _captured_output(span_exporter) == REFUSAL_MESSAGE + + async def test_a_check_blocked_on_output_records_the_masked_response( + self, capturing_iorails, call_log, httpx_mock, span_exporter + ): + """An output block behind the mask still records the checked response as the mask left it.""" + _wire(capturing_iorails, call_log, httpx_mock, OUTPUT_CONTENT_SAFETY_BLOCKS) + + await capturing_iorails.check_async(CHECKED_CONVERSATION, rail_types=[RailType.OUTPUT]) + + assert _captured_input(span_exporter) == [ + {"role": "user", "content": USER_INPUT}, + {"role": "assistant", "content": MASKED_MAIN_OUTPUT}, + ] + assert _captured_output(span_exporter) == REFUSAL_MESSAGE + + async def test_a_blocked_generation_records_the_masked_user_message( + self, capturing_iorails, call_log, httpx_mock, span_exporter + ): + """Generation keeps the mask on the request span when an input rail behind it refuses.""" + _wire(capturing_iorails, call_log, httpx_mock, CONTENT_SAFETY_BLOCKS) + + await capturing_iorails.generate_async(messages=[{"role": "user", "content": USER_INPUT}]) + + assert _captured_input(span_exporter) == [{"role": "user", "content": MASKED_INPUT}] + assert _captured_output(span_exporter) == REFUSAL_MESSAGE + + async def test_a_blocked_stream_records_the_masked_user_message(self, call_log, httpx_mock, span_exporter): + """Streaming keeps the mask on the request span when an input rail behind it refuses.""" + async with _capturing(_capturing_streaming_config(), span_exporter) as engine: + _wire(engine, call_log, httpx_mock, CONTENT_SAFETY_BLOCKS) + streamed = await _streamed(engine) + + assert _captured_input(span_exporter) == [{"role": "user", "content": MASKED_INPUT}] + assert "content_blocked" in streamed From 0d91150536ce40b8ff25c9c7d39ad8aa0ef05b58 Mon Sep 17 00:00:00 2001 From: tgasser-nv <200644301+tgasser-nv@users.noreply.github.com> Date: Fri, 2 Oct 2026 11:12:37 -0500 Subject: [PATCH 4/9] Check in tests (expect to fail) --- tests/guardrails/test_guardrails_types.py | 6 ++++ tests/guardrails/test_iorails_check.py | 26 ++++++++++++++ tests/guardrails/test_rails_manager.py | 43 +++++++++++++++++++++++ 3 files changed, 75 insertions(+) diff --git a/tests/guardrails/test_guardrails_types.py b/tests/guardrails/test_guardrails_types.py index 465934f37f..ac3668794f 100644 --- a/tests/guardrails/test_guardrails_types.py +++ b/tests/guardrails/test_guardrails_types.py @@ -143,6 +143,12 @@ def test_records_stay_out_of_equality(self): assert RailResult.block(reason="blocked", records=(record,)) == RailResult.block(reason="blocked") + def test_a_rewrite_before_a_block_stays_out_of_equality(self): + """What the rails rewrote ahead of a block is capture data, so the block alone is the verdict compared.""" + kept = RailResult(RailOutcome.block(reason="blocked"), rewrite_before_block="my ssn is ") + + assert kept == RailResult.block(reason="blocked") + def test_is_unhashable_and_says_so(self): """The wrapped outcome is unhashable, and the error names this type rather than leaking from inside.""" with pytest.raises(TypeError, match="unhashable type: 'RailResult'"): diff --git a/tests/guardrails/test_iorails_check.py b/tests/guardrails/test_iorails_check.py index 729915d63d..5a746dfcfe 100644 --- a/tests/guardrails/test_iorails_check.py +++ b/tests/guardrails/test_iorails_check.py @@ -22,6 +22,7 @@ import asyncio import logging +from dataclasses import replace from unittest.mock import AsyncMock, MagicMock, patch import pytest @@ -918,6 +919,31 @@ async def test_an_input_mask_is_captured_masked_when_the_output_blocks(self, ior assert capture.call_args.args[1][0] == {"role": "user", "content": MASKED_USER_TEXT} + async def test_an_input_mask_is_captured_masked_when_the_input_blocks(self, iorails): + """A block behind the input mask still captures the user message as the mask left it.""" + iorails._content_capture_enabled = True + blocked = replace(_unsafe("content safety check input"), rewrite_before_block=MASKED_USER_TEXT) + iorails.rails_manager.is_input_safe = AsyncMock(return_value=blocked) + + with patch("nemoguardrails.guardrails.iorails.set_request_content") as capture: + await iorails.check_async([{"role": "user", "content": USER_TEXT}]) + + assert capture.call_args.args[1] == [{"role": "user", "content": MASKED_USER_TEXT}] + + async def test_an_output_mask_is_captured_masked_when_the_output_blocks(self, iorails): + """A block behind the output mask still captures the checked response as the mask left it.""" + iorails._content_capture_enabled = True + blocked = replace(_unsafe("content safety check output"), rewrite_before_block=MASKED_BOT_TEXT) + iorails.rails_manager.is_output_safe = AsyncMock(return_value=blocked) + + with patch("nemoguardrails.guardrails.iorails.set_request_content") as capture: + await iorails.check_async(CONVERSATION, rail_types=[RailType.OUTPUT]) + + assert capture.call_args.args[1] == [ + {"role": "user", "content": USER_TEXT}, + {"role": "assistant", "content": MASKED_BOT_TEXT}, + ] + async def test_an_unmasked_check_is_captured_as_it_arrived(self, iorails): """With no rewrite, the span records the messages the caller sent.""" iorails._content_capture_enabled = True diff --git a/tests/guardrails/test_rails_manager.py b/tests/guardrails/test_rails_manager.py index b880804ea9..40208bed15 100644 --- a/tests/guardrails/test_rails_manager.py +++ b/tests/guardrails/test_rails_manager.py @@ -1888,6 +1888,49 @@ async def test_a_block_behind_a_rewrite_returns_the_block(self, nemoguards_rails assert result.triggered_rail == "topic safety check input" assert len(result.records) == 2 + async def test_a_block_behind_a_rewrite_keeps_the_rewrite_for_the_record(self, nemoguards_rails_manager): + """The block is the verdict, but the masked text still reaches whoever records the request.""" + self._install_input_pair( + nemoguards_rails_manager, + StubRail(_mask_user_message(MASKED)), + StubRail(RailOutcome.block(reason="off topic")), + ) + + result = await nemoguards_rails_manager.is_input_safe(SSN_MESSAGES, enabled=INPUT_PAIR) + + assert result.rewrite_before_block == MASKED + + async def test_an_output_block_behind_a_rewrite_keeps_the_rewritten_response(self, nemoguards_rails_manager): + """The output direction keeps the response as the rails ahead of the block rewrote it.""" + second_flow = "mask pii on output" + self._install( + nemoguards_rails_manager, + RailDirection.OUTPUT, + { + CONTENT_SAFETY_OUTPUT_FLOW: StubRail(_mask_bot_message("call me on ")), + second_flow: StubRail(RailOutcome.block(reason="unsafe")), + }, + ) + + result = await nemoguards_rails_manager._run_rails_sequential( + [CONTENT_SAFETY_OUTPUT_FLOW, second_flow], RailDirection.OUTPUT, SSN_MESSAGES, "call me on 555-0100" + ) + + assert result.is_safe is False + assert result.rewrite_before_block == "call me on " + + async def test_a_block_with_no_rewrite_ahead_keeps_nothing(self, nemoguards_rails_manager): + """With nothing rewritten before the block, there is no rewrite for the record to keep.""" + self._install_input_pair( + nemoguards_rails_manager, + StubRail(RailOutcome.allow()), + StubRail(RailOutcome.block(reason="off topic")), + ) + + result = await nemoguards_rails_manager.is_input_safe(SSN_MESSAGES, enabled=INPUT_PAIR) + + assert result.rewrite_before_block is None + async def test_a_rewrite_leaves_the_callers_messages_untouched(self, nemoguards_rails_manager): """The caller's list arrives by identity, so masking must not edit the conversation it owns.""" messages = [{"role": "user", "content": "my ssn is 123-45-6789"}] From 5bf119f86e47e7428ba204751d39e716074b8cf5 Mon Sep 17 00:00:00 2001 From: tgasser-nv <200644301+tgasser-nv@users.noreply.github.com> Date: Fri, 2 Oct 2026 11:12:47 -0500 Subject: [PATCH 5/9] Check in fixes --- .../observability/tracing/content-capture.mdx | 9 ++++-- .../using-python-apis/check-messages.mdx | 2 +- nemoguardrails/guardrails/guardrails_types.py | 6 +++- nemoguardrails/guardrails/iorails.py | 31 ++++++++++++++++--- nemoguardrails/guardrails/rails_manager.py | 14 ++++++++- 5 files changed, 52 insertions(+), 10 deletions(-) diff --git a/docs/observability/tracing/content-capture.mdx b/docs/observability/tracing/content-capture.mdx index c9c327e726..bedd7e8b5d 100644 --- a/docs/observability/tracing/content-capture.mdx +++ b/docs/observability/tracing/content-capture.mdx @@ -44,7 +44,7 @@ Sampling parameters (`gen_ai.request.temperature`, `max_tokens`, and so on) and A rails-only [`check_async()`](/run-guardrailed-inference/using-python-apis/check-messages) uses the same capture behavior. The checked messages are recorded in `guardrails.request.input`, and the returned content is recorded in `guardrails.request.output`. -When an input or output rail rewrites the text it checked, `guardrails.request.input` carries the rewritten messages. +When an input or output rail rewrites the text it checked, `guardrails.request.input` carries the rewritten messages, even when a later rail blocks the check. Every rail that runs records its own `guardrails.rail` attributes. A check makes no main-model call, so it produces no CLIENT span for the protected model. `IORails` captures LLM calls from rail actions as usual. @@ -230,9 +230,12 @@ In both cases, if no content was produced, IORails omits the output rather than - Content capture is **off by default**. Prompts and responses are never written to spans unless you explicitly enable it. - Captured content can include PII or other sensitive data. Treat your tracing backend as a system that stores user content, and apply the same access controls and retention limits you would apply to any store of conversation data. -- Masking keeps text out of `guardrails.request.input`, but not out of the `guardrails.rail` spans, which record the input each rail received. - The masking rail's own span carries the text before the mask. +- Masking rails limit what the request span records, not what the whole trace records. + `guardrails.request.input` records the turn a masking rail checked as the rail rewrote it, even when a later rail blocks the request. + Earlier turns of the conversation, which no rail checks, are recorded as the caller sent them. + Each `guardrails.rail` span records the input its rail received, so the masking rail's own span carries the text before the mask. On a `check_async()` call, whose conversation includes the checked response, every rail that runs records that response as it was sent. + On `generate_async()` and `stream_async()`, the protected model's `chat ` CLIENT span records its response before an output rail masks it. - Use `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=false` to force capture off across every service from the environment, regardless of what individual configs request. This is useful as a deployment-wide guardrail for regulated environments. - The [OpenTelemetry GenAI semantic conventions](https://opentelemetry.io/docs/specs/semconv/gen-ai/) recommend against capturing message content by default because of these privacy risks. diff --git a/docs/run-rails/using-python-apis/check-messages.mdx b/docs/run-rails/using-python-apis/check-messages.mdx index bdfd43509c..91c9d5a2c6 100644 --- a/docs/run-rails/using-python-apis/check-messages.mdx +++ b/docs/run-rails/using-python-apis/check-messages.mdx @@ -427,7 +427,7 @@ When your application requires that a rail actually ran, assert that the message A check rejected by a full admission queue increments `guardrails.nonstream.rejections`. For more information, refer to [Metric Reference](/observability/metrics/reference). - Content capture records the checked messages in `guardrails.request.input` and the returned content in `guardrails.request.output` on the check's request span. - When an input or output rail rewrites the text it checked, for example to mask personal data, the span records the rewritten messages rather than the ones the caller sent. + When an input or output rail rewrites the text it checked, for example to mask personal data, the span records the rewritten messages rather than the ones the caller sent, even when a later rail blocks the check. For more information, refer to [Capturing Prompt and Response Content](/observability/tracing/content-capture). The synchronous `check()` builds a short-lived engine with tracing and metrics disabled, so it emits no spans and no metrics. Use `check_async()` on instrumented paths. diff --git a/nemoguardrails/guardrails/guardrails_types.py b/nemoguardrails/guardrails/guardrails_types.py index a7c102e5bb..833dd4c312 100644 --- a/nemoguardrails/guardrails/guardrails_types.py +++ b/nemoguardrails/guardrails/guardrails_types.py @@ -103,7 +103,8 @@ class RailResult: than a second copy that could drift. What this type adds is the aggregation ``RailOutcome`` has no concept of, because it belongs to running *many* rails: which one blocked (``triggered_rail``), which tool calls or results it blocked - (``tool_violations``) and what every rail did (``records``). + (``tool_violations``), what every rail did (``records``), and what the rails ahead + of a block rewrote the checked text to (``rewrite_before_block``). ``records`` carries the per-rail execution records for every rail that ran in this check (not just the blocking one), so IORails can synthesize a ``GenerationLog``. @@ -119,6 +120,9 @@ class RailResult: triggered_rail: str | None = None tool_violations: tuple[ToolViolation, ...] = () records: tuple[RailCallRecord, ...] = field(default=(), compare=False) + # Capture data like ``records``, not part of the verdict: the block stands, but a record of + # the request that carried the text a mask removed would undo the mask. + rewrite_before_block: str | None = field(default=None, compare=False) __hash__ = None @property diff --git a/nemoguardrails/guardrails/iorails.py b/nemoguardrails/guardrails/iorails.py index 1f54a133ef..4b958547a5 100644 --- a/nemoguardrails/guardrails/iorails.py +++ b/nemoguardrails/guardrails/iorails.py @@ -602,6 +602,20 @@ def _rewrite_last_assistant_message(messages: LLMMessages, text: str) -> LLMMess raise ValueError("no assistant message to rewrite") +def _apply_input_rewrite_before_block(messages: LLMMessages, blocked: RailResult) -> LLMMessages: + """Return *messages* with the user turn as the input rails rewrote it before one of them blocked.""" + if blocked.rewrite_before_block is None: + return messages + return rewrite_user_message(messages, blocked.rewrite_before_block) + + +def _apply_output_rewrite_before_block(messages: LLMMessages, blocked: RailResult) -> LLMMessages: + """Return *messages* with the checked response as the output rails rewrote it before one of them blocked.""" + if blocked.rewrite_before_block is None: + return messages + return _rewrite_last_assistant_message(messages, blocked.rewrite_before_block) + + def _blocked_message(result: RailResult) -> str: """The text a blocked turn returns, which says whether the rail broke or fired.""" if result.failed: @@ -642,7 +656,7 @@ class _TurnConversation: @dataclass(frozen=True, slots=True) class _GeneratedTurn: - """The main model's response and the messages it read, or no response when the rails blocked. + """The main model's response and the messages it read, or on a block no response and the messages the rails left. ``blocked_by`` carries the verdict that stopped the turn, because the caller renders the message from it and a rail that broke reads differently from one that fired. @@ -1142,6 +1156,9 @@ def _blocked_return(blocked_by: RailResult) -> Union[LLMMessage, GenerationRespo messages, req_id, llm_kwargs, input_enabled=input_enabled, records_out=records ) + # Before the blocked return, so a block behind a mask still records the masked messages. + if conversation is not None: + conversation.messages = turn.messages if turn.blocked_by is not None: return _blocked_return(turn.blocked_by) @@ -1153,8 +1170,6 @@ def _blocked_return(blocked_by: RailResult) -> Union[LLMMessage, GenerationRespo # generation bug to the caller as a guardrail decision. raise RuntimeError("generation returned no response without naming a blocking rail") messages = turn.messages - if conversation is not None: - conversation.messages = messages # Log raw content before reasoning extraction and think-token removal log.debug("[%s] Raw LLM response: %s", req_id, truncate(response.content)) @@ -1245,7 +1260,8 @@ async def _do_generate_sequential( log.info("[%s] Input blocked: %s", req_id, display_reason(input_result)) if self._metrics_enabled: record_request_blocked(RailDirection.INPUT) - return _GeneratedTurn(response=None, messages=messages, blocked_by=input_result) + rewritten_messages = _apply_input_rewrite_before_block(messages, input_result) + return _GeneratedTurn(response=None, messages=rewritten_messages, blocked_by=input_result) rewritten = _rewritten_user_message(input_result) if rewritten is not None: @@ -1547,6 +1563,8 @@ async def _do_check( log.info("[%s] Input blocked: %s", req_id, display_reason(input_result)) if self._metrics_enabled: record_request_blocked(RailDirection.INPUT) + if conversation is not None: + conversation.messages = _apply_input_rewrite_before_block(messages, input_result) return _blocked_check_result(input_result) rewritten = _rewritten_user_message(input_result) if rewritten is not None: @@ -1575,6 +1593,8 @@ async def _do_check( log.info("[%s] Output blocked: %s", req_id, display_reason(output_result)) if self._metrics_enabled: record_request_blocked(RailDirection.OUTPUT) + if conversation is not None: + conversation.messages = _apply_output_rewrite_before_block(messages, output_result) return _blocked_check_result(output_result) rewritten = _rewritten_bot_message(output_result) if rewritten is not None: @@ -1753,6 +1773,9 @@ async def _generation_task(request_span): log.info("[%s] Input blocked: %s", req_id, display_reason(input_result)) if self._metrics_enabled: record_request_blocked(RailDirection.INPUT) + # The request span records ``messages`` once the stream ends, so a mask applied + # ahead of the block has to reach it before the caller can see the stream close. + messages = _apply_input_rewrite_before_block(messages, input_result) await streaming_handler.push_chunk( self._guardrails_violation_payload( f"Blocked by input rails: {client_reason(input_result)}", "input_rails" diff --git a/nemoguardrails/guardrails/rails_manager.py b/nemoguardrails/guardrails/rails_manager.py index 79f0b39630..53f5ef3de6 100644 --- a/nemoguardrails/guardrails/rails_manager.py +++ b/nemoguardrails/guardrails/rails_manager.py @@ -263,6 +263,18 @@ def _result_after_rewrites( return RailResult(RailOutcome.transform([(_REWRITABLE_TARGET[direction], final_text)]), records=records) +def _blocked_after_rewrites( + blocked: RailResult, + original_text: str, + final_text: str, + records: tuple[RailCallRecord, ...], +) -> RailResult: + """Keep the block as the verdict, and what the rails ahead of it rewrote, so a record of the request keeps a mask.""" + if final_text == original_text: + return replace(blocked, records=records) + return replace(blocked, records=records, rewrite_before_block=final_text) + + def _model_free_record( flow: str, rail_type: str, result: RailResult, tool_name: Optional[str] = None ) -> RailCallRecord: @@ -962,7 +974,7 @@ async def _run_rails_sequential( log.debug("[%s] %s flow %s result %s", req_id, direction.value, flow, result) if not result.is_safe: log.info("[%s] %s flow %s blocked", req_id, direction.value, flow) - return replace(result, records=tuple(collected)) + return _blocked_after_rewrites(result, original_text, final_text, tuple(collected)) if result.outcome.is_transform: final_text = _rewritten_text(result.outcome, direction, flow) log.info("[%s] %s flow %s rewrote the text it checked", req_id, direction.value, flow) From e1484ca92f48554069034885d44c0f34e9c6e135 Mon Sep 17 00:00:00 2001 From: tgasser-nv <200644301+tgasser-nv@users.noreply.github.com> Date: Mon, 5 Oct 2026 10:57:17 -0500 Subject: [PATCH 6/9] Add test (expected to fail) --- tests/guardrails/test_rails_manager.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/tests/guardrails/test_rails_manager.py b/tests/guardrails/test_rails_manager.py index 40208bed15..bbdf579640 100644 --- a/tests/guardrails/test_rails_manager.py +++ b/tests/guardrails/test_rails_manager.py @@ -2542,3 +2542,17 @@ async def test_the_rails_that_ran_are_still_recorded(self, nemoguards_rails_mana ) assert len(result.records) == 1 + + async def test_the_rewrite_already_applied_is_kept_for_the_record(self, nemoguards_rails_manager): + """The block keeps the text the rails ahead of it left, so the record of the request keeps their mask.""" + nemoguards_rails_manager._rails[(RailDirection.INPUT, CONTENT_SAFETY_INPUT_FLOW)] = StubRail( + _mask_user_message("") + ) + nemoguards_rails_manager._rails[(RailDirection.INPUT, TOPIC_SAFETY_INPUT_FLOW)] = StubRail( + _mask_user_message(MASKED) + ) + + result = await nemoguards_rails_manager.is_input_safe(SSN_MESSAGES, enabled=INPUT_PAIR) + + assert result.is_safe is False + assert result.rewrite_before_block == "" From 9ee12f3653b62ff9169000704f1c074777f65ece Mon Sep 17 00:00:00 2001 From: tgasser-nv <200644301+tgasser-nv@users.noreply.github.com> Date: Mon, 5 Oct 2026 11:03:01 -0500 Subject: [PATCH 7/9] Add fix to keep applied rewrite on a missing-turn block --- docs/observability/tracing/content-capture.mdx | 2 ++ nemoguardrails/guardrails/rails_manager.py | 11 ++++++----- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/docs/observability/tracing/content-capture.mdx b/docs/observability/tracing/content-capture.mdx index bedd7e8b5d..6ee9841dc8 100644 --- a/docs/observability/tracing/content-capture.mdx +++ b/docs/observability/tracing/content-capture.mdx @@ -233,7 +233,9 @@ In both cases, if no content was produced, IORails omits the output rather than - Masking rails limit what the request span records, not what the whole trace records. `guardrails.request.input` records the turn a masking rail checked as the rail rewrote it, even when a later rail blocks the request. Earlier turns of the conversation, which no rail checks, are recorded as the caller sent them. + When no mask reaches the request, `guardrails.request.input` records the turn as the caller sent it: the masking rail fails, a tool-result rail blocks before the input rails run, or a stream closes before its input rails finish. Each `guardrails.rail` span records the input its rail received, so the masking rail's own span carries the text before the mask. + A rail that calls an LLM also records the prompt it built from that input on the call's own `chat ` CLIENT span. On a `check_async()` call, whose conversation includes the checked response, every rail that runs records that response as it was sent. On `generate_async()` and `stream_async()`, the protected model's `chat ` CLIENT span records its response before an output rail masks it. - Use `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=false` to force capture off across every service from the environment, regardless of what individual configs request. This is useful as a deployment-wide guardrail for regulated environments. diff --git a/nemoguardrails/guardrails/rails_manager.py b/nemoguardrails/guardrails/rails_manager.py index 53f5ef3de6..b4ad5128e4 100644 --- a/nemoguardrails/guardrails/rails_manager.py +++ b/nemoguardrails/guardrails/rails_manager.py @@ -976,24 +976,25 @@ async def _run_rails_sequential( log.info("[%s] %s flow %s blocked", req_id, direction.value, flow) return _blocked_after_rewrites(result, original_text, final_text, tuple(collected)) if result.outcome.is_transform: - final_text = _rewritten_text(result.outcome, direction, flow) + rewritten_text = _rewritten_text(result.outcome, direction, flow) log.info("[%s] %s flow %s rewrote the text it checked", req_id, direction.value, flow) if direction is RailDirection.INPUT: try: - messages = rewrite_user_message(messages, final_text) + messages = rewrite_user_message(messages, rewritten_text) except ValueError: # Blocking keeps a misbehaving rail inside the fail-closed envelope, # rather than failing the request as a server error. log.error( "[%s] %s flow %s rewrote a turn this request does not have", req_id, direction.value, flow ) - return RailResult.block( + blocked = RailResult.block( reason="a rail rewrote a message this request does not have", triggered_rail=_get_flow_name(flow) or flow, - records=tuple(collected), ) + return _blocked_after_rewrites(blocked, original_text, final_text, tuple(collected)) else: - bot_response = final_text + bot_response = rewritten_text + final_text = rewritten_text return _result_after_rewrites(direction, original_text, final_text, tuple(collected)) async def _run_tool_rails_sequential( From 4b89177b18f47dd80740a5e439f0bfa51bbed0ee Mon Sep 17 00:00:00 2001 From: tgasser-nv <200644301+tgasser-nv@users.noreply.github.com> Date: Mon, 5 Oct 2026 16:35:05 -0500 Subject: [PATCH 8/9] Tighten capture wirnig and document unmasked check responses --- docs/observability/tracing/content-capture.mdx | 7 ++++--- .../using-python-apis/check-messages.mdx | 1 + nemoguardrails/guardrails/iorails.py | 18 +++++++----------- nemoguardrails/guardrails/telemetry.py | 2 +- 4 files changed, 13 insertions(+), 15 deletions(-) diff --git a/docs/observability/tracing/content-capture.mdx b/docs/observability/tracing/content-capture.mdx index 6ee9841dc8..fee9308b75 100644 --- a/docs/observability/tracing/content-capture.mdx +++ b/docs/observability/tracing/content-capture.mdx @@ -232,11 +232,12 @@ In both cases, if no content was produced, IORails omits the output rather than - Captured content can include PII or other sensitive data. Treat your tracing backend as a system that stores user content, and apply the same access controls and retention limits you would apply to any store of conversation data. - Masking rails limit what the request span records, not what the whole trace records. `guardrails.request.input` records the turn a masking rail checked as the rail rewrote it, even when a later rail blocks the request. - Earlier turns of the conversation, which no rail checks, are recorded as the caller sent them. - When no mask reaches the request, `guardrails.request.input` records the turn as the caller sent it: the masking rail fails, a tool-result rail blocks before the input rails run, or a stream closes before its input rails finish. + It records earlier turns of the conversation, which no rail checks, as the caller sent them. + When no mask reaches the request, it records the turn as the caller sent it: the masking rail fails, a tool-result rail blocks before the input rails run, or a stream closes or fails before its input rails finish. + On a `check_async()` call, it records the checked response as the caller sent it when no output rail runs to mask it: an input rail blocks first, or `rail_types` leaves out `RailType.OUTPUT`. Each `guardrails.rail` span records the input its rail received, so the masking rail's own span carries the text before the mask. A rail that calls an LLM also records the prompt it built from that input on the call's own `chat ` CLIENT span. - On a `check_async()` call, whose conversation includes the checked response, every rail that runs records that response as it was sent. + On a `check_async()` call, whose conversation includes the checked response, every rail that runs records that response as the caller sent it. On `generate_async()` and `stream_async()`, the protected model's `chat ` CLIENT span records its response before an output rail masks it. - Use `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=false` to force capture off across every service from the environment, regardless of what individual configs request. This is useful as a deployment-wide guardrail for regulated environments. - The [OpenTelemetry GenAI semantic conventions](https://opentelemetry.io/docs/specs/semconv/gen-ai/) recommend against capturing message content by default because of these privacy risks. diff --git a/docs/run-rails/using-python-apis/check-messages.mdx b/docs/run-rails/using-python-apis/check-messages.mdx index 91c9d5a2c6..740942ffe1 100644 --- a/docs/run-rails/using-python-apis/check-messages.mdx +++ b/docs/run-rails/using-python-apis/check-messages.mdx @@ -428,6 +428,7 @@ When your application requires that a rail actually ran, assert that the message For more information, refer to [Metric Reference](/observability/metrics/reference). - Content capture records the checked messages in `guardrails.request.input` and the returned content in `guardrails.request.output` on the check's request span. When an input or output rail rewrites the text it checked, for example to mask personal data, the span records the rewritten messages rather than the ones the caller sent, even when a later rail blocks the check. + An output rail that never runs masks nothing: when an input rail blocks first, or `rail_types` leaves out `RailType.OUTPUT`, the span records the checked response as the caller sent it. For more information, refer to [Capturing Prompt and Response Content](/observability/tracing/content-capture). The synchronous `check()` builds a short-lived engine with tracing and metrics disabled, so it emits no spans and no metrics. Use `check_async()` on instrumented paths. diff --git a/nemoguardrails/guardrails/iorails.py b/nemoguardrails/guardrails/iorails.py index 4b958547a5..9e60cc0080 100644 --- a/nemoguardrails/guardrails/iorails.py +++ b/nemoguardrails/guardrails/iorails.py @@ -1070,8 +1070,8 @@ async def _run_generate( log.error("[%s] generate_async failed time=%.1fms", req_id, elapsed_ms, exc_info=True) raise # Captured at the traced_request boundary, so a future early-return in _do_generate - # is covered. The messages are the ones the model read: a span carrying the text a - # mask removed would defeat the mask. + # is covered. The messages are the ones the rails left, which the model read unless a + # rail blocked: a span carrying the text a mask removed would defeat the mask. if self._content_capture_enabled: set_request_content(request_span, conversation.messages, _response_content_for_capture(result)) elapsed_ms = (time.monotonic() - t0) * 1000 @@ -1521,7 +1521,7 @@ async def _do_check( req_id: str, *, tools: Optional[list[dict]] = None, - conversation: Optional["_TurnConversation"] = None, + conversation: "_TurnConversation", ) -> RailsResult: """Core check pipeline: run the requested input, output and tool rails on messages.""" log.info("[%s] check called", req_id) @@ -1563,15 +1563,13 @@ async def _do_check( log.info("[%s] Input blocked: %s", req_id, display_reason(input_result)) if self._metrics_enabled: record_request_blocked(RailDirection.INPUT) - if conversation is not None: - conversation.messages = _apply_input_rewrite_before_block(messages, input_result) + conversation.messages = _apply_input_rewrite_before_block(messages, input_result) return _blocked_check_result(input_result) rewritten = _rewritten_user_message(input_result) if rewritten is not None: log.info("[%s] Input rails rewrote the user message", req_id) messages = rewrite_user_message(messages, rewritten) - if conversation is not None: - conversation.messages = messages + conversation.messages = messages if not reports_output: pass_content = rewritten else: @@ -1593,16 +1591,14 @@ async def _do_check( log.info("[%s] Output blocked: %s", req_id, display_reason(output_result)) if self._metrics_enabled: record_request_blocked(RailDirection.OUTPUT) - if conversation is not None: - conversation.messages = _apply_output_rewrite_before_block(messages, output_result) + conversation.messages = _apply_output_rewrite_before_block(messages, output_result) return _blocked_check_result(output_result) rewritten = _rewritten_bot_message(output_result) if rewritten is not None: log.info("[%s] Output rails rewrote the response", req_id) pass_content = rewritten messages = _rewrite_last_assistant_message(messages, rewritten) - if conversation is not None: - conversation.messages = messages + conversation.messages = messages else: log.info("[%s] Output rails requested but no assistant content to check; skipping", req_id) diff --git a/nemoguardrails/guardrails/telemetry.py b/nemoguardrails/guardrails/telemetry.py index f3e6d8acc1..21ecad0d9f 100644 --- a/nemoguardrails/guardrails/telemetry.py +++ b/nemoguardrails/guardrails/telemetry.py @@ -701,7 +701,7 @@ def set_request_content( input_messages: LLMMessages, output_text: Optional[str] = None, ) -> None: - """Capture caller-facing input/output on the ``guardrails.request`` SERVER span. + """Capture the request's input/output on the ``guardrails.request`` SERVER span. Uses ``guardrails.request.input`` (JSON-encoded input messages) and ``guardrails.request.output`` (the text actually returned to the caller) From d50b0732c01f605b67aec9a83d4566734d823212 Mon Sep 17 00:00:00 2001 From: tgasser-nv <200644301+tgasser-nv@users.noreply.github.com> Date: Wed, 7 Oct 2026 13:42:46 -0500 Subject: [PATCH 9/9] Document that any rail blocking before the output rails leaves the checked response unmasked Signed-off-by: tgasser-nv <200644301+tgasser-nv@users.noreply.github.com> --- docs/observability/tracing/content-capture.mdx | 2 +- docs/run-rails/using-python-apis/check-messages.mdx | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/observability/tracing/content-capture.mdx b/docs/observability/tracing/content-capture.mdx index fee9308b75..2edd1a7344 100644 --- a/docs/observability/tracing/content-capture.mdx +++ b/docs/observability/tracing/content-capture.mdx @@ -234,7 +234,7 @@ In both cases, if no content was produced, IORails omits the output rather than `guardrails.request.input` records the turn a masking rail checked as the rail rewrote it, even when a later rail blocks the request. It records earlier turns of the conversation, which no rail checks, as the caller sent them. When no mask reaches the request, it records the turn as the caller sent it: the masking rail fails, a tool-result rail blocks before the input rails run, or a stream closes or fails before its input rails finish. - On a `check_async()` call, it records the checked response as the caller sent it when no output rail runs to mask it: an input rail blocks first, or `rail_types` leaves out `RailType.OUTPUT`. + On a `check_async()` call, it records the checked response as the caller sent it when no output rail runs to mask it: another rail blocks before the output rails run, or `rail_types` leaves out `RailType.OUTPUT`. Each `guardrails.rail` span records the input its rail received, so the masking rail's own span carries the text before the mask. A rail that calls an LLM also records the prompt it built from that input on the call's own `chat ` CLIENT span. On a `check_async()` call, whose conversation includes the checked response, every rail that runs records that response as the caller sent it. diff --git a/docs/run-rails/using-python-apis/check-messages.mdx b/docs/run-rails/using-python-apis/check-messages.mdx index 740942ffe1..6dc8ceae4d 100644 --- a/docs/run-rails/using-python-apis/check-messages.mdx +++ b/docs/run-rails/using-python-apis/check-messages.mdx @@ -428,7 +428,7 @@ When your application requires that a rail actually ran, assert that the message For more information, refer to [Metric Reference](/observability/metrics/reference). - Content capture records the checked messages in `guardrails.request.input` and the returned content in `guardrails.request.output` on the check's request span. When an input or output rail rewrites the text it checked, for example to mask personal data, the span records the rewritten messages rather than the ones the caller sent, even when a later rail blocks the check. - An output rail that never runs masks nothing: when an input rail blocks first, or `rail_types` leaves out `RailType.OUTPUT`, the span records the checked response as the caller sent it. + An output rail that never runs masks nothing: when another rail blocks before the output rails run, or `rail_types` leaves out `RailType.OUTPUT`, the span records the checked response as the caller sent it. For more information, refer to [Capturing Prompt and Response Content](/observability/tracing/content-capture). The synchronous `check()` builds a short-lived engine with tracing and metrics disabled, so it emits no spans and no metrics. Use `check_async()` on instrumented paths.