agent-control-plane / agents /action_draft.py
edangx100's picture
Update app and control-plane modules to latest
fbe9dad
Raw
History Blame Contribute Delete
6.3 kB
"""LLM-facing action drafts — the JSON-string ``arguments`` workaround.
This module isolates the fix for the *empty-`arguments` from the live model*
problem. The real :class:`~control_plane.schema.ProposedAction` types
``arguments`` as a ``dict[str, Any]``, which compiles to a *property-less* JSON
schema (``{"type": "object"}``). Models won't invent keys for a schema that names
none, so the live model returns ``arguments={}`` and every argument-taking action
fails. Models DO reliably fill a string field, so the agent emits these "draft"
types where ``arguments`` is a JSON *string*, and :func:`draft_to_decision` parses
that string back into a genuine ``ProposedAction``.
The fix is contained entirely in the agent layer: ``ProposedAction``, the gate,
grants, and the backends are untouched — they still receive a normal ``arguments``
dict.
Why a separate module: the drafts are a serialization DTO shaped around a
provider limitation, distinct from the governed loop itself. Keeping them here
lets :mod:`agents.soc_agent` stay focused on the loop, and the draft↔decision
conversion (plus its single error type) be tested in isolation.
"""
from __future__ import annotations
import json
from typing import Any
from pydantic import BaseModel, ConfigDict, Field, ValidationError
from control_plane.schema import (
AgentDecision,
AutonomyTier,
Backend,
ProposedAction,
RiskLevel,
)
class MalformedAgentOutput(Exception):
"""Raised by a brain when the model's output cannot be validated into a turn.
Pydantic-AI already retries the model on a schema violation (local
validation/repair). This is raised only when that repair is *exhausted*, so
the harness can surface it as an error result rather than letting it crash
the run ("malformed output triggers local validation/repair, not a
crash"). :func:`draft_to_decision` raises the same error for bad
``arguments_json`` so the loop has a single failure type to handle.
"""
class ProposedActionDraft(BaseModel):
"""LLM-facing twin of :class:`~control_plane.schema.ProposedAction` whose
``arguments`` is a JSON string the model can actually fill."""
model_config = ConfigDict(extra="forbid")
# incident_id and agent_id are *context the harness owns*, not choices for the
# model: the run already knows which incident it is about and which identity is
# signing. They default to empty and are injected by draft_to_decision, so a
# model that omits them (or returns "") can't break the run. Kept in the schema
# (rather than removed) so older drafts that do send them still validate.
incident_id: str = Field(
default="", description="Set by the system — leave blank."
)
agent_id: str = Field(
default="", description="Set by the system — leave blank."
)
autonomy_tier: AutonomyTier
backend: Backend
action_name: str
arguments_json: str = Field(
default="{}",
description=(
"The action's arguments as a JSON object string, e.g. "
'\'{"alert_id": "ALERT-2001"}\'. Use "{}" when the action needs none.'
),
)
reason: str
evidence: list[str] = Field(default_factory=list)
class AgentDecisionDraft(BaseModel):
"""LLM-facing twin of :class:`~control_plane.schema.AgentDecision`."""
model_config = ConfigDict(extra="forbid")
summary: str
risk_level: RiskLevel
proposed_action: ProposedActionDraft
prohibited_actions_detected: list[str] = Field(default_factory=list)
def draft_to_decision(
draft: AgentDecisionDraft,
*,
incident_id: str | None = None,
agent_id: str | None = None,
) -> AgentDecision:
"""Convert a model-emitted draft into a real, validated :class:`AgentDecision`.
Parses the JSON-string arguments back into a dict, then builds the genuine
schema object (which re-runs all of ``ProposedAction``'s validation). Any bad
JSON, a non-object payload, or a schema violation is raised as
:class:`MalformedAgentOutput` so the harness surfaces it as a clean error
instead of crashing the run.
``incident_id`` and ``agent_id`` are *injected by the harness from run context*
when supplied — they are not the model's to choose: the run already knows which
incident it concerns and which identity is signing (and the signature must match
that identity). Injecting them means a model that leaves these blank can no
longer abort the run, and the proposal can never claim a different identity than
the one that signs it. When not supplied, the draft's own values are used
(keeping older callers working).
"""
pa = draft.proposed_action
try:
arguments: Any = json.loads(pa.arguments_json.strip() or "{}")
except json.JSONDecodeError as exc:
raise MalformedAgentOutput(f"arguments_json is not valid JSON: {exc}") from exc
if not isinstance(arguments, dict):
raise MalformedAgentOutput("arguments_json must be a JSON object")
# Harness-owned context wins over whatever the model put (or left blank).
resolved_incident = incident_id if incident_id else pa.incident_id
resolved_agent = agent_id if agent_id else pa.agent_id
# Normalize the action name: models sometimes echo a signature-style name like
# "get_asset(asset_id)" instead of the bare "get_asset" — which is not a real
# action and would be denied. Strip anything from the first "(" so the bare verb
# survives; the actual arguments come from arguments_json, not the name.
resolved_action = pa.action_name.split("(", 1)[0].strip()
try:
return AgentDecision(
summary=draft.summary,
risk_level=draft.risk_level,
proposed_action=ProposedAction(
incident_id=resolved_incident,
agent_id=resolved_agent,
autonomy_tier=pa.autonomy_tier,
backend=pa.backend,
action_name=resolved_action,
arguments=arguments,
reason=pa.reason,
evidence=pa.evidence,
),
prohibited_actions_detected=draft.prohibited_actions_detected,
)
except ValidationError as exc:
raise MalformedAgentOutput(str(exc)) from exc