From 49fd2921b7c495d8fc3e35208d4c07380fa759b4 Mon Sep 17 00:00:00 2001 From: Ling-Sen Peng Date: Tue, 6 Oct 2026 16:31:27 -0700 Subject: [PATCH] Add ServerGuardrail for guardrails the server enforces on an agent's LLM calls --- examples/agents/120_server_guardrails.py | 64 +++++++++++++++++++ src/conductor/ai/agents/__init__.py | 2 + src/conductor/ai/agents/agent.py | 12 +++- src/conductor/ai/agents/config_serializer.py | 2 + src/conductor/ai/agents/guardrail.py | 66 ++++++++++++++++++++ tests/unit/ai/test_config_serializer.py | 31 +++++++++ 6 files changed, 176 insertions(+), 1 deletion(-) create mode 100644 examples/agents/120_server_guardrails.py diff --git a/examples/agents/120_server_guardrails.py b/examples/agents/120_server_guardrails.py new file mode 100644 index 00000000..20c64d6f --- /dev/null +++ b/examples/agents/120_server_guardrails.py @@ -0,0 +1,64 @@ +"""Server guardrails — guardrails the Conductor server enforces on the agent's LLM calls. + +``ServerGuardrail`` names a guardrail defined on the server (or a builtin such as +``"pii"``). The server checks it inside every LLM task the agent compiles, on the +prompt before it reaches the model and on the model's answer. Nothing runs in this +process, and sub-agents compiled with the agent are covered too. + +Server guardrails sit in the same ``guardrails`` list as the SDK's own kinds: + +- ``ServerGuardrail("pii", action="REDACT")`` — the server redacts personal data, + so the email address in the prompt never reaches the model +- ``ServerGuardrail("secrets", action="BLOCK")`` — the LLM task fails if an API key + or credential appears +- ``RegexGuardrail`` — unchanged: compiled into the agent's own retry loop + +A server that does not enforce server guardrails rejects the agent rather than run +it unguarded. + +Requirements: + - Orkes Conductor server with guardrails enabled + - CONDUCTOR_SERVER_URL=http://localhost:8080/api as environment variable + - CONDUCTOR_AGENT_LLM_MODEL=openai/gpt-4o-mini as environment variable +""" + +from conductor.ai.agents import Agent, AgentRuntime, OnFail, RegexGuardrail, ServerGuardrail +from settings import settings + +agent = Agent( + name="support_replier", + model=settings.llm_model, + instructions="Draft a short, friendly reply to the customer's message.", + guardrails=[ + ServerGuardrail("pii", action="REDACT"), + ServerGuardrail("secrets", action="BLOCK"), + RegexGuardrail( + name="no_internal_links", + patterns=[r"https?://[\w.-]*\.internal\b"], + message="Do not include internal links in the reply.", + on_fail=OnFail.RETRY, + ), + ], +) + +# REDACT: the email address is removed before the prompt reaches the model; the run completes. +prompt = ( + "Hi, I'm Jane Doe (jane.doe@example.com). My last invoice charged me twice. " + "Can you help?" +) + +# BLOCK: an API key in the message fails the LLM task, so the run fails instead of sending it. +block_prompt = ( + "Our integration stopped working. Here is the key we use: api_FAKEKEY0123456789EXAMPLE. " + "Can you check what is wrong?" +) + +if __name__ == "__main__": + with AgentRuntime() as runtime: + print("── REDACT ──") + runtime.run(agent, prompt).print_result() + + print("── BLOCK ──") + blocked = runtime.run(agent, block_prompt) + print(f"Status: {blocked.status}") + print(f"Error: {blocked.error}") diff --git a/src/conductor/ai/agents/__init__.py b/src/conductor/ai/agents/__init__.py index fce5b308..fb9c6356 100644 --- a/src/conductor/ai/agents/__init__.py +++ b/src/conductor/ai/agents/__init__.py @@ -70,6 +70,7 @@ def get_weather(city: str) -> str: OnFail, Position, RegexGuardrail, + ServerGuardrail, guardrail, ) @@ -307,6 +308,7 @@ def resolve_credentials(task: object, names: list) -> dict: "Position", "RegexGuardrail", "LLMGuardrail", + "ServerGuardrail", # Termination conditions "TerminationCondition", "TerminationResult", diff --git a/src/conductor/ai/agents/agent.py b/src/conductor/ai/agents/agent.py index b199aa11..92828c9a 100644 --- a/src/conductor/ai/agents/agent.py +++ b/src/conductor/ai/agents/agent.py @@ -719,7 +719,17 @@ def __init__( self.strategy = strategy self.router = router self.output_type = output_type - self.guardrails: List[Any] = list(guardrails) if guardrails else [] + from conductor.ai.agents.guardrail import ServerGuardrail + + all_guardrails = list(guardrails) if guardrails else [] + self.guardrails: List[Any] = [ + g for g in all_guardrails if not isinstance(g, ServerGuardrail) + ] + # Enforced by the server inside the agent's LLM tasks, so kept apart from the guardrails + # this runtime compiles or runs as workers. + self.task_guardrails: List[ServerGuardrail] = [ + g for g in all_guardrails if isinstance(g, ServerGuardrail) + ] self.memory = memory self.dependencies: Dict[str, Any] = dict(dependencies) if dependencies else {} self.max_turns = max_turns diff --git a/src/conductor/ai/agents/config_serializer.py b/src/conductor/ai/agents/config_serializer.py index fcc8ed67..5179a150 100644 --- a/src/conductor/ai/agents/config_serializer.py +++ b/src/conductor/ai/agents/config_serializer.py @@ -126,6 +126,8 @@ def _serialize_agent(self, agent: "Agent") -> dict: # Guardrails if agent.guardrails: config["guardrails"] = [self._serialize_guardrail(g) for g in agent.guardrails] + if getattr(agent, "task_guardrails", None): + config["taskGuardrails"] = [g.to_binding() for g in agent.task_guardrails] # Memory if hasattr(agent, "memory") and agent.memory: diff --git a/src/conductor/ai/agents/guardrail.py b/src/conductor/ai/agents/guardrail.py index ef50e604..b708b9be 100644 --- a/src/conductor/ai/agents/guardrail.py +++ b/src/conductor/ai/agents/guardrail.py @@ -410,3 +410,69 @@ def __repr__(self) -> str: return ( f"LLMGuardrail(name={self.name!r}, model={self._model!r}, position={self.position!r})" ) + + +class ServerGuardrail: + """A guardrail defined on the Conductor server, bound by name to the agent's LLM calls. + + The server enforces it inside every LLM task the agent compiles, including those of + sub-agents compiled with it, before the prompt reaches the model and on the model's answer. + Nothing runs in this process. A server that does not enforce server guardrails rejects + the agent rather than run it unguarded. + + Args: + name: The guardrail's name on the server, or a builtin such as ``"pii"``. + at: Where to check: ``"PROMPT"``, ``"USER_MESSAGE"`` or ``"MODEL_OUTPUT"``. + Defaults to every point the guardrail can check. + action: ``"REDACT"``, ``"BLOCK"``, ``"REASK"`` or ``"HUMAN"``. + Defaults to the guardrail's own default action. + version: Pin a version. Defaults to the latest. + on_error: ``"BLOCK"`` or ``"RETRY_THEN_BLOCK"`` when the check itself fails. + max_attempts: Re-asks allowed when ``action="REASK"``. + on_exhausted: The action once re-asks run out. + + Example:: + + agent = Agent( + name="support", + model="openai/gpt-4o", + guardrails=[ServerGuardrail("pii", action="REDACT"), ServerGuardrail("support.no-refunds")], + ) + """ + + def __init__( + self, + name: str, + *, + at: Optional[str] = None, + action: Optional[str] = None, + version: Optional[int] = None, + on_error: Optional[str] = None, + max_attempts: Optional[int] = None, + on_exhausted: Optional[str] = None, + ) -> None: + if not name: + raise ValueError("ServerGuardrail requires a name") + self.name = name + self.at = at + self.action = action + self.version = version + self.on_error = on_error + self.max_attempts = max_attempts + self.on_exhausted = on_exhausted + + def to_binding(self) -> Union[str, dict]: + """The entry for an LLM task's ``guardrails`` input: the name alone when nothing is tuned.""" + options = { + "at": self.at, + "action": self.action, + "version": self.version, + "onError": self.on_error, + "maxAttempts": self.max_attempts, + "onExhausted": self.on_exhausted, + } + options = {key: value for key, value in options.items() if value is not None} + return {"guardrail": self.name, **options} if options else self.name + + def __repr__(self) -> str: + return f"ServerGuardrail(name={self.name!r})" diff --git a/tests/unit/ai/test_config_serializer.py b/tests/unit/ai/test_config_serializer.py index 56097d6b..5cbf7614 100644 --- a/tests/unit/ai/test_config_serializer.py +++ b/tests/unit/ai/test_config_serializer.py @@ -101,6 +101,37 @@ def test_serialize_guardrails_regex(self): assert g["mode"] == "block" assert g["onFail"] == "retry" + def test_serialize_server_guardrails(self): + """ServerGuardrail entries serialize as taskGuardrails bindings, apart from guardrails.""" + from conductor.ai.agents.agent import Agent + from conductor.ai.agents.guardrail import RegexGuardrail, ServerGuardrail + + regex = RegexGuardrail(patterns=[r"\d{3}-\d{2}-\d{4}"], name="no_ssn") + agent = Agent( + name="test", + model="openai/gpt-4o", + guardrails=[ + ServerGuardrail("pii"), + regex, + ServerGuardrail("support.no-refunds", at="MODEL_OUTPUT", action="REASK", max_attempts=2), + ], + ) + config = self.serializer.serialize(agent) + + assert agent.guardrails == [regex] + assert [g["name"] for g in config["guardrails"]] == ["no_ssn"] + assert config["taskGuardrails"] == [ + "pii", + {"guardrail": "support.no-refunds", "at": "MODEL_OUTPUT", "action": "REASK", "maxAttempts": 2}, + ] + + def test_serialize_without_server_guardrails_has_no_task_guardrails(self): + from conductor.ai.agents.agent import Agent + + config = self.serializer.serialize(Agent(name="test", model="openai/gpt-4o")) + + assert "taskGuardrails" not in config + def test_serialize_guardrails_llm(self): """LLMGuardrail serializes with model and policy.""" from conductor.ai.agents.agent import Agent