Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion evaluators/contrib/galileo/Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@ sync:
uv sync

test:
uv run --with pytest --with pytest-asyncio --with pytest-cov --package $(PACKAGE) pytest tests --cov=src --cov-report=xml:../../../coverage-evaluators-galileo.xml -q
uv run --extra dev --with pytest --with pytest-asyncio --with pytest-cov --package $(PACKAGE) pytest tests --cov=src --cov-report=xml:../../../coverage-evaluators-galileo.xml -q

lint:
uv run --with ruff --package $(PACKAGE) ruff check --config ../../../pyproject.toml src/
Expand Down
6 changes: 6 additions & 0 deletions evaluators/contrib/galileo/pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,9 @@ dependencies = [

[project.optional-dependencies]
dev = [
"agent-control-sdk>=8.11.0",
"agent-control-telemetry>=8.6.0",
"agent-control-engine>=8.6.0",
"pytest>=8.0.0",
"pytest-asyncio>=0.23.0",
"pytest-cov>=4.0.0",
Expand All @@ -37,3 +40,6 @@ packages = ["src/agent_control_evaluator_galileo"]
[tool.uv.sources]
agent-control-evaluators = { path = "../../builtin", editable = true }
agent-control-models = { path = "../../../models", editable = true }
agent-control-sdk = { path = "../../../sdks/python", editable = true }
agent-control-telemetry = { path = "../../../telemetry", editable = true }
agent-control-engine = { path = "../../../engine", editable = true }
Original file line number Diff line number Diff line change
@@ -0,0 +1,58 @@
"""Cross-package check: StepRecorder trees feed the Galileo record factory."""

from __future__ import annotations

from agent_control import record_step
from agent_control_evaluator_galileo.records.factory import record_from_step as _record_from_step


def test_recorder_built_trace_matches_demo_shape():
"""A trace with one llm/tool/retriever child each builds a 3-span Trace."""

def policy_search(query: str) -> dict:
return {"docs": ["policy-1"]}

def account_lookup(account_id: str) -> dict:
return {"balance": 500}

def banking_llm(prompt: str) -> str:
return f"response: {prompt}"

with record_step("trace", "banking_trace", input={"request": "r1"}) as trace:
trace.call(
policy_search, query="refund policy", step_type="retriever", step_name="policy_search"
)
trace.call(
account_lookup, account_id="acct-1", step_type="tool", step_name="account_lookup"
)
trace.call(banking_llm, "draft a reply", step_type="llm", step_name="banking_llm")
trace.output = {"status": "done"}

record = _record_from_step(trace.build())

assert type(record).__name__ == "Trace"
assert len(record.spans) == 3
assert {type(span).__name__ for span in record.spans} == {
"LlmSpan",
"ToolSpan",
"RetrieverSpan",
}


def test_recorder_built_session_matches_demo_shape():
"""A session with 2 traces, each with >=2 spans, builds the matching Session."""
with record_step("session", "banking_session") as session:
for i in range(2):
with session.child("trace", f"turn_{i}", input={"turn": i}) as trace:
with trace.child("llm", "respond", input="hi") as llm:
llm.output = "hello"
with trace.child("tool", "lookup", input={}) as tool:
tool.output = {}
trace.output = {"turn": i, "status": "ok"}

record = _record_from_step(session.build())

assert type(record).__name__ == "Session"
assert len(record.traces) == 2
for sub_trace in record.traces:
assert len(sub_trace.spans) >= 2
54 changes: 54 additions & 0 deletions sdks/python/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -84,6 +84,60 @@ async with agent_control.AgentControlClient() as client:
The existing `evaluate_controls` helper remains available for callers that
prefer its field-based convenience arguments.

## Building a step with children

A control's scope can target any step type, including one that aggregates
several already-executed child steps - `Step.children` just has to be
populated by the caller. Building that tree by hand means hand-rolling a
side-channel record for every leaf call and converting it into `Step`
objects afterward:

```python
# Before: a hand-built dict record next to the real return value
def execute_policy_search(query):
docs = search_policy_documents(query)
record = {"type": "lookup", "query": query, "docs": docs}
return StepExecution(value=docs, record=record)

children = []
result = execute_policy_search(query)
children.append(result.record)
...
workflow_step = build_agent_control_step({"type": "workflow", "children": children, ...})
```

`agent_control.record_step()` builds the same tree incrementally, in `Step`
vocabulary, with no intermediate dict and no converter:

```python
with agent_control.record_step("workflow", "banking_workflow", input={"request": req}) as workflow:
with workflow.child("lookup", "policy_lookup", input=query) as span:
span.output = search_policy_documents(query)

account = workflow.call(lookup_account, account_id="acct-1001", step_type="tool")
plan = workflow.call(run_banking_model, req, step_type="llm", tools=TOOL_DEFINITIONS)
workflow.output = {"status": "planned", "message": plan["content"]}

result = await workflow.evaluate(stage="post")
```

`workflow.call(...)`/`await workflow.acall(...)` run the function and record
it as a child using the same capture logic as `@control()` (input from bound
arguments, output from the return value), then return the real result -
removing the need for a separate `StepExecution`-style wrapper. A failed
call is still recorded (with the error in `context`) before the exception is
re-raised. `workflow.child(...)` nests another recorder the same way, so a
parent step can nest children of any depth, e.g. `session.child("trace",
...)` for callers that model multi-turn sessions of traces.

`workflow.build()` produces the frozen `Step`, with `children=[]` instead of
omitted (`None`) for a step type passed via `container_types` even when no
children were recorded - e.g. `record_step("workflow", ..., container_types=
{"workflow"})` for a workflow that legitimately ran with zero children.
`workflow.evaluate(...)` builds the step and evaluates it in one call via
`evaluate_step()` - the same function `evaluate_controls()` uses internally
once its `Step` is built.

## Sharing an OpenTelemetry provider with Google ADK

When Google ADK and Agent Control should export through the same OpenTelemetry
Expand Down
7 changes: 6 additions & 1 deletion sdks/python/src/agent_control/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -92,7 +92,7 @@ async def handle_input(user_message: str) -> str:
)
from .client import AgentControlClient
from .control_decorators import ControlSteerError, ControlViolationError, control
from .evaluation import check_evaluation_with_local, evaluate_controls
from .evaluation import check_evaluation_with_local, evaluate_controls, evaluate_step
from .observability import (
LogConfig,
add_event,
Expand All @@ -116,6 +116,7 @@ async def handle_input(user_message: str) -> str:
)
from .otel_sink import control_event_to_otel_span
from .runtime_auth import validate_http_field_name
from .step_recorder import StepRecorder, record_step
from .tracing import (
get_current_span_id,
get_current_trace_id,
Expand Down Expand Up @@ -1619,6 +1620,10 @@ async def main():
# Local evaluation
"check_evaluation_with_local",
"evaluate_controls",
"evaluate_step",
# Step recorder (incremental trace/session Step tree builder)
"record_step",
"StepRecorder",
# Tracing
"get_trace_and_span_ids",
"get_current_trace_id",
Expand Down
90 changes: 63 additions & 27 deletions sdks/python/src/agent_control/evaluation.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
"""Evaluation check operations for Agent Control SDK."""

from collections.abc import Awaitable, Callable
from collections.abc import Awaitable, Callable, Mapping, Sequence
from dataclasses import dataclass
from inspect import iscoroutinefunction
from typing import Any, Literal, cast
Expand Down Expand Up @@ -515,6 +515,58 @@ def _with_parse_errors(result: EvaluationResult) -> EvaluationResult:
return _with_parse_errors(EvaluationResult(is_safe=True, confidence=1.0))


async def evaluate_step(
step: Step,
*,
agent_name: str,
stage: Literal["pre", "post"] = "pre",
target_type: str | None = None,
target_id: str | None = None,
trace_id: str | None = None,
span_id: str | None = None,
) -> EvaluationResult:
"""Evaluate controls for an already-built ``Step``.

This is the shared tail of :func:`evaluate_controls`: resolve the
session target, open a client, and run local/server evaluation. Use
it directly when the ``Step`` - including any ``children`` - was
already assembled, e.g. via :class:`~agent_control.step_recorder.StepRecorder`.

When ``target_type`` and ``target_id`` are both supplied, the request
is target-bearing: the server merges target bindings into the
effective control set. If they are omitted, the SDK falls back to the
target context fixed at ``init()`` time when present. A per-call
override that disagrees with the session target is rejected because
the cached controls were fetched for the session target and would
otherwise drive stale local-first evaluation.
"""
if state.server_url is None:
raise RuntimeError("Server URL not configured. Call agent_control.init() first.")

target_type, target_id = _resolve_session_target(target_type, target_id)
resolved_controls = state.server_controls or []

async with AgentControlClient(
base_url=state.server_url,
api_key=state.api_key,
api_key_header=state.api_key_header,
runtime_token_header=state.runtime_token_header,
runtime_token_cache=state.runtime_token_cache,
) as client:
return await check_evaluation_with_local(
client=client,
agent_name=agent_name,
step=step,
stage=stage,
controls=resolved_controls,
target_type=target_type,
target_id=target_id,
trace_id=trace_id,
span_id=span_id,
event_agent_name=agent_name,
)


async def evaluate_controls(
step_name: str,
*,
Expand All @@ -523,7 +575,7 @@ async def evaluate_controls(
context: dict[str, Any] | None = None,
tools: list[dict[str, JSONValue]] | None = None,
ground_truth: JSONValue | None = None,
children: list[Step] | None = None,
children: Sequence[Step | Mapping[str, Any]] | None = None,
step_type: str = "llm",
stage: Literal["pre", "post"] = "pre",
agent_name: str,
Expand All @@ -544,11 +596,6 @@ async def evaluate_controls(
"""
step_type = ensure_step_type(step_type)

if state.server_url is None:
raise RuntimeError("Server URL not configured. Call agent_control.init() first.")

target_type, target_id = _resolve_session_target(target_type, target_id)

default_value = {} if step_type == "tool" else ""
step_dict: dict[str, Any] = {
"type": step_type,
Expand All @@ -566,24 +613,13 @@ async def evaluate_controls(
step_dict["children"] = children

step_obj = Step(**step_dict) # type: ignore[arg-type]
resolved_controls = state.server_controls or []

async with AgentControlClient(
base_url=state.server_url,
api_key=state.api_key,
api_key_header=state.api_key_header,
runtime_token_header=state.runtime_token_header,
runtime_token_cache=state.runtime_token_cache,
) as client:
return await check_evaluation_with_local(
client=client,
agent_name=agent_name,
step=step_obj,
stage=stage,
controls=resolved_controls,
target_type=target_type,
target_id=target_id,
trace_id=trace_id,
span_id=span_id,
event_agent_name=agent_name,
)
return await evaluate_step(
step_obj,
agent_name=agent_name,
stage=stage,
target_type=target_type,
target_id=target_id,
trace_id=trace_id,
span_id=span_id,
)
Loading
Loading