Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
38 changes: 35 additions & 3 deletions controller/src/reconciler/runtime.rs
Original file line number Diff line number Diff line change
Expand Up @@ -375,9 +375,10 @@ pub fn build_runtime_plan(
}
RuntimeKind::SemanticKernel => Err(RuntimePlanError::AdapterMissing("SemanticKernel")),
RuntimeKind::LangGraph => {
// Phase H#2: LangGraph Python adapter wired end-to-end.
// TypeScript flavour is gated as ShapeInvalid until the
// TS adapter image ships (mirrors the MAF .NET strategy).
// LangGraph adapters wired end-to-end for both Python
// (`runtimes/langgraph/`) and TypeScript / Node 22
// (`runtimes/langgraph-ts/`). `plan_langgraph` matches on
// `language` and selects the matching image.
let cfg = runtime.lang_graph.as_ref().expect("validated above");
plan_langgraph(cfg)
}
Expand Down Expand Up @@ -684,6 +685,17 @@ mod tests {
MafLanguage, MicrosoftAgentFrameworkConfig, OciAgentCode, OpenAIAgentsConfig,
OpenClawConfig, PydanticAiConfig, SemanticKernelConfig,
};
use std::sync::Mutex;

/// Serialises tests that mutate process-wide image-override env vars
/// (e.g. `OPENAI_AGENTS_RUNTIME_IMAGE`). Without this guard, the
/// default multi-threaded test harness lets one test's `set_var`
/// leak into another test's `*_default_image()` call, producing
/// flaky failures of the shape:
/// left: "myacr.azurecr.io/openai-agents:pinned"
/// right: "azureclawacr.azurecr.io/azureclaw-runtime-openai-agents:latest"
/// All env-mutating tests in this module take the lock at the top.
static ENV_LOCK: Mutex<()> = Mutex::new(());

fn rt_openclaw(image: Option<&str>) -> RuntimeSpec {
RuntimeSpec {
Expand Down Expand Up @@ -912,6 +924,7 @@ mod tests {

#[test]
fn anthropic_default_image_falls_back_when_env_unset() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
// SAFETY: this test does not run in parallel with env writers
// (no other test sets ANTHROPIC_RUNTIME_IMAGE).
unsafe {
Expand Down Expand Up @@ -998,6 +1011,7 @@ mod tests {

#[test]
fn langgraph_default_image_falls_back_when_env_unset() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("LANGGRAPH_RUNTIME_IMAGE");
}
Expand Down Expand Up @@ -1038,6 +1052,7 @@ mod tests {

#[test]
fn plan_langgraph_typescript_dispatches_to_ts_image() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("LANGGRAPH_TS_RUNTIME_IMAGE");
}
Expand Down Expand Up @@ -1094,6 +1109,7 @@ mod tests {

#[test]
fn pydantic_ai_default_image_falls_back_when_env_unset() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("PYDANTIC_AI_RUNTIME_IMAGE");
}
Expand Down Expand Up @@ -1280,6 +1296,7 @@ mod tests {

#[test]
fn plan_openai_agents_uses_default_adapter_image_when_env_unset() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
// Defensive: clear the env var so the test isn't influenced by
// an operator-side override leaking from the dev shell.
// SAFETY: serial-test is not on this crate; tests in the same
Expand Down Expand Up @@ -1307,6 +1324,7 @@ mod tests {

#[test]
fn plan_openai_agents_passes_through_python_version_and_extra_env() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("OPENAI_AGENTS_RUNTIME_IMAGE");
}
Expand All @@ -1332,6 +1350,7 @@ mod tests {

#[test]
fn plan_openai_agents_user_extra_env_overrides_python_version_key_when_explicitly_set() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
// Defensive: if a user explicitly puts RUNTIME_PYTHON_VERSION in
// extra_env, their value wins over the producer's
// python_version-derived default. The merge order
Expand All @@ -1358,6 +1377,7 @@ mod tests {

#[test]
fn plan_openai_agents_carries_user_entrypoint_into_command() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("OPENAI_AGENTS_RUNTIME_IMAGE");
}
Expand All @@ -1380,6 +1400,7 @@ mod tests {

#[test]
fn plan_openai_agents_propagates_agent_code() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("OPENAI_AGENTS_RUNTIME_IMAGE");
}
Expand All @@ -1402,6 +1423,7 @@ mod tests {

#[test]
fn openai_agents_default_image_honours_env_override() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::set_var(
"OPENAI_AGENTS_RUNTIME_IMAGE",
Expand All @@ -1419,6 +1441,7 @@ mod tests {

#[test]
fn openai_agents_default_image_treats_blank_env_as_unset() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::set_var("OPENAI_AGENTS_RUNTIME_IMAGE", " ");
}
Expand All @@ -1431,6 +1454,7 @@ mod tests {

#[test]
fn build_runtime_plan_dispatches_openai_agents_to_producer() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("OPENAI_AGENTS_RUNTIME_IMAGE");
}
Expand Down Expand Up @@ -1459,6 +1483,7 @@ mod tests {

#[test]
fn plan_maf_uses_default_python_image_when_env_unset() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("MAF_RUNTIME_IMAGE");
}
Expand All @@ -1481,6 +1506,7 @@ mod tests {

#[test]
fn plan_maf_explicit_python_language_succeeds() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("MAF_RUNTIME_IMAGE");
}
Expand All @@ -1494,6 +1520,7 @@ mod tests {

#[test]
fn plan_maf_default_language_is_python() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("MAF_RUNTIME_IMAGE");
}
Expand All @@ -1510,6 +1537,7 @@ mod tests {

#[test]
fn plan_maf_passes_entrypoint_and_extra_env() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("MAF_RUNTIME_IMAGE");
}
Expand Down Expand Up @@ -1547,6 +1575,7 @@ mod tests {

#[test]
fn plan_maf_user_extra_env_overrides_controller_default() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("MAF_RUNTIME_IMAGE");
}
Expand All @@ -1573,6 +1602,7 @@ mod tests {

#[test]
fn maf_python_default_image_honours_env_override() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::set_var("MAF_RUNTIME_IMAGE", "myacr.azurecr.io/maf-python:pinned");
}
Expand All @@ -1586,6 +1616,7 @@ mod tests {

#[test]
fn maf_python_default_image_treats_blank_env_as_unset() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::set_var("MAF_RUNTIME_IMAGE", " ");
}
Expand All @@ -1598,6 +1629,7 @@ mod tests {

#[test]
fn build_runtime_plan_dispatches_maf_python_to_producer() {
let _g = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
unsafe {
std::env::remove_var("MAF_RUNTIME_IMAGE");
}
Expand Down
33 changes: 33 additions & 0 deletions examples/byo-quickstart/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -76,3 +76,36 @@ Tear down with:
```bash
kubectl delete clawsandbox byo-quickstart
```

## 5. Verify strict-mode admission (recommended for production)

The example above runs cleanly under either `byoStrict=false` (default
— Phase 2 behaviour, advisory warnings only) or `byoStrict=true`
(Phase 3 — rejection at admission time). For production we recommend
the latter. To see strict mode actually reject a malformed CR, use the
intentionally-invalid demo manifest:

```bash
# Roll the controller with strict mode on:
helm upgrade azureclaw deploy/helm/azureclaw \
--reuse-values --set controller.byoStrict=true
kubectl rollout status deploy/azureclaw-controller -n azureclaw-system

# Apply a CR with a bogus contractVersion:
kubectl apply -f k8s/clawsandbox-strict-demo.yaml

# Confirm the controller refused:
kubectl get clawsandbox byo-strict-demo -n azureclaw-system \
-o jsonpath='{.status.conditions}' | jq
# Expect: type=Degraded, status=True, reason=BYOContractInvalid

# No Deployment / Service / NetworkPolicy should exist:
kubectl get all -n azureclaw-byo-strict-demo 2>/dev/null || echo "namespace empty (as expected)"

# Tear down:
kubectl delete -f k8s/clawsandbox-strict-demo.yaml
```

See [`docs/operations/byo-strict.md`](../../docs/operations/byo-strict.md)
for the full list of CR-level checks and the Phase 4 roadmap for
registry-side label introspection.
51 changes: 51 additions & 0 deletions examples/byo-quickstart/k8s/clawsandbox-strict-demo.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
# Strict-mode demo: this CR is INTENTIONALLY invalid.
#
# When the controller is rolled out with `controller.byoStrict=true`,
# applying this manifest must NOT produce a half-rendered Deployment.
# Instead the ClawSandbox is stamped:
#
# status.conditions:
# - type: Degraded
# status: "True"
# reason: BYOContractInvalid
# message: "byo.contractVersion=`v999` is not in the supported set ..."
#
# Use this to verify strict-mode admission end-to-end before deploying
# real workloads. With `byoStrict=false` (default) the same manifest is
# accepted with a `WARN` advisory in controller logs — strict mode is
# what gives the rejection teeth.
---
apiVersion: azureclaw.azure.com/v1alpha1
kind: InferencePolicy
metadata:
name: byo-strict-demo
namespace: azureclaw-system
labels:
azureclaw.azure.com/sandbox: byo-strict-demo
spec:
appliesTo:
sandboxName: byo-strict-demo
modelPreference:
primary:
provider: azure-openai
deployment: gpt-4.1
tokenBudget:
perRequestMax: 4000
perDayMax: 200000
---
apiVersion: azureclaw.azure.com/v1alpha1
kind: ClawSandbox
metadata:
name: byo-strict-demo
namespace: azureclaw-system
spec:
runtime:
kind: BYO
byo:
# Intentionally bogus contract version. Strict mode rejects.
image: ghcr.io/example/byo-quickstart:v1
contractVersion: v999
sandbox:
isolation: standard
inferenceRef:
name: byo-strict-demo
22 changes: 13 additions & 9 deletions mesh-plugin/src/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -420,15 +420,19 @@ export function definePluginEntry() {
name: "mesh_inbox",
description:
"Read messages received via the E2E encrypted AGT mesh. ALWAYS call " +
"this tool when the user asks about inbox, replies, or peer messages — " +
"do NOT rely on what you remember from earlier turns, because new " +
"messages may have arrived since then. Default behaviour is *peek-only* " +
"and shows only entries you haven't read yet; messages stay in the " +
"inbox so you can re-read them. Pass mark_read=true once you have " +
"acted on the contents to flag them as seen, or unread_only=false to " +
"also see entries from previous turns. The response includes a " +
"`diagnostics` block with lifecycle counters so you can tell apart " +
"'never received' from 'already consumed by an offload waiter'.",
"this tool FIRST whenever your task description says a peer agent " +
"has sent you data, output, or a reply — peer messages always " +
"arrive before you process them, so checking the inbox is your " +
"default opening move whenever you're told to consume something " +
"from another agent. Do NOT rely on what you remember from earlier " +
"turns, because new messages may have arrived since then. Default " +
"behaviour is *peek-only* and shows only entries you haven't read " +
"yet; messages stay in the inbox so you can re-read them. Pass " +
"mark_read=true once you have acted on the contents to flag them " +
"as seen, or unread_only=false to also see entries from previous " +
"turns. The response includes a `diagnostics` block with lifecycle " +
"counters so you can tell apart 'never received' from 'already " +
"consumed by an offload waiter'.",
parameters: {
type: "object",
properties: {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,7 @@
receive_messages,
send_message,
)
from azureclaw_runtime_anthropic.mesh_tools import build_mesh_tools
from azureclaw_runtime_anthropic.otel import init_telemetry
from azureclaw_runtime_anthropic.runtime import bootstrap

Expand All @@ -35,6 +36,7 @@
"PLATFORM_MCP_URL",
"__version__",
"bootstrap",
"build_mesh_tools",
"get_token",
"init_telemetry",
"receive_messages",
Expand Down
73 changes: 73 additions & 0 deletions runtimes/anthropic/src/azureclaw_runtime_anthropic/mesh_tools.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,73 @@
# Copyright (c) Microsoft Corporation.
# Licensed under the MIT License.
"""
First-class AgentMesh tools for the Anthropic Claude Agent SDK.

Wraps :mod:`mesh` as ``claude_agent_sdk.tool``-decorated callables. See
``runtimes/openai-agents/.../mesh_tools.py`` for the rationale.
"""

from __future__ import annotations

import json
import logging
from typing import Any, List

from azureclaw_runtime_anthropic.mesh import receive_messages, send_message

logger = logging.getLogger(__name__)


_MESH_INBOX_DESCRIPTION = (
"Drain pending AgentMesh messages addressed to this agent. "
"ALWAYS call this tool FIRST whenever your task description says a "
"peer agent has sent you data, when you were just resumed after a "
"handoff, or when you suspect a parent agent is waiting for a reply. "
"Returns a JSON array of message envelopes; an empty array means no "
"pending messages. Calling this is cheap and idempotent — when in "
"doubt, call it."
)

_MESH_SEND_DESCRIPTION = (
"Send an AgentMesh A2A TaskEnvelope to a peer agent identified by "
"their DID (e.g. ``did:mesh:<name>``). Use this to delegate work to a "
"sub-agent or reply to a parent agent. Returns the relay's "
"acknowledgement."
)


def build_mesh_tools() -> List[Any]:
"""Return ``claude_agent_sdk`` tool definitions."""
from claude_agent_sdk import tool # type: ignore

@tool(
name="mesh_inbox",
description=_MESH_INBOX_DESCRIPTION,
input_schema={"type": "object", "properties": {}, "required": []},
)
async def mesh_inbox(_args: dict) -> dict:
msgs = receive_messages()
return {"content": [{"type": "text", "text": json.dumps(msgs)}]}

@tool(
name="mesh_send",
description=_MESH_SEND_DESCRIPTION,
input_schema={
"type": "object",
"properties": {
"target_agent": {"type": "string"},
"content": {"type": "string"},
"skill_id": {"type": "string", "default": "chat"},
},
"required": ["target_agent", "content"],
},
)
async def mesh_send(args: dict) -> dict:
ack = send_message(
args["target_agent"],
args["content"],
skill_id=args.get("skill_id", "chat"),
)
return {"content": [{"type": "text", "text": json.dumps(ack)}]}

return [mesh_inbox, mesh_send]
Loading
Loading