kiln_ai.synthetic_user

Local synthetic-user player.

Plays the synthetic-user side of a conversation locally, calling the LLM with the user's own provider keys.

Public surface:

 1"""Local synthetic-user player.
 2
 3Plays the synthetic-user side of a conversation locally, calling the LLM
 4with the user's own provider keys.
 5
 6Public surface:
 7
 8- `SyntheticUserDriver` — construct once per case, call `respond()` per turn.
 9- `SyntheticUserInfo` / `SyntheticUserDriverConfig` — typed configs.
10- `SyntheticUserCase` — input contract for the multi-turn drive loop.
11- `parse_synthetic_user_info` — tagged-blob parser.
12- `SyntheticUserInfoParseError` — raised on malformed blob.
13- `role_swap` — exposed for callers that drive the loop themselves.
14- `drive_case_for_eval` — transient one-case drive for the eval runner.
15"""
16
17from kiln_ai.synthetic_user.case import SyntheticUserCase
18from kiln_ai.synthetic_user.driver import SyntheticUserDriver
19from kiln_ai.synthetic_user.eval_drive import drive_case_for_eval
20from kiln_ai.synthetic_user.models import (
21    SyntheticUserDriverConfig,
22    SyntheticUserInfo,
23)
24from kiln_ai.synthetic_user.parser import (
25    SyntheticUserInfoParseError,
26    parse_synthetic_user_info,
27)
28from kiln_ai.synthetic_user.role_swap import role_swap
29
30__all__ = [
31    "SyntheticUserCase",
32    "SyntheticUserDriver",
33    "SyntheticUserDriverConfig",
34    "SyntheticUserInfo",
35    "SyntheticUserInfoParseError",
36    "drive_case_for_eval",
37    "parse_synthetic_user_info",
38    "role_swap",
39]
class SyntheticUserCase(pydantic.main.BaseModel):
15class SyntheticUserCase(BaseModel):
16    """One case for the multi-turn SU drive loop.
17
18    `seed_prompt` is the first user-side message sent into the target
19    task. `synthetic_user_info` is the persona/goal/behavior_guidance
20    blob, parsed by the caller into the typed SyntheticUserInfo the
21    driver builds the SU's system prompt from.
22    """
23
24    seed_prompt: str = Field(..., min_length=1)
25    synthetic_user_info: str = Field(..., min_length=1)

One case for the multi-turn SU drive loop.

seed_prompt is the first user-side message sent into the target task. synthetic_user_info is the persona/goal/behavior_guidance blob, parsed by the caller into the typed SyntheticUserInfo the driver builds the SU's system prompt from.

seed_prompt: str
synthetic_user_info: str
model_config: ClassVar[pydantic.config.ConfigDict] = {}

Configuration for the model, should be a dictionary conforming to [ConfigDict][pydantic.config.ConfigDict].

class SyntheticUserDriver:
 43class SyntheticUserDriver:
 44    """Plays one synthetic user across multiple turns.
 45
 46    Constructed once per case from the typed persona + driver config;
 47    `respond(conversation)` is called once per turn. Callers holding the
 48    tagged wire blob parse it first (kiln_ai.synthetic_user.parser) — blob
 49    parsing stays at the wire boundary. The adapter is built at
 50    construction time and reused across all turns of the case.
 51    """
 52
 53    def __init__(
 54        self,
 55        synthetic_user_info: SyntheticUserInfo,
 56        driver_config: SyntheticUserDriverConfig,
 57    ):
 58        self._info = synthetic_user_info
 59        self._driver_config = driver_config
 60        self._system_prompt: str = render_system_prompt(self._info)
 61
 62        # In-memory Task; nothing is persisted. The persona-playing system
 63        # prompt rides on `prior_trace[0]` each call, so this `instruction`
 64        # is effectively unused at runtime — kiln_ai's MultiturnFormatter
 65        # uses the first system message in `prior_trace` and skips the
 66        # task's instruction when `prior_trace` is non-empty. The Task
 67        # model requires a non-empty instruction, hence the placeholder.
 68        self._task = Task(
 69            name="synthetic_user_driver",
 70            description="In-memory SU player. Not persisted.",
 71            instruction=(
 72                "Placeholder — the persona-playing system prompt is supplied "
 73                "via prior_trace on every adapter call."
 74            ),
 75        )
 76        # Same RunConfigProperties shape used elsewhere; `structured_output_mode`
 77        # is `default` because SU output is free text (no JSON schema).
 78        self._run_config = KilnAgentRunConfigProperties(
 79            model_name=driver_config.model_name,
 80            model_provider_name=driver_config.model_provider_name,
 81            prompt_id="simple_prompt_builder",
 82            structured_output_mode=StructuredOutputMode.default,
 83            tools_config=ToolsRunConfig(tools=[]),
 84        )
 85        self._adapter = adapter_for_task(self._task, self._run_config)
 86
 87    async def respond(
 88        self, conversation: list[ChatCompletionMessageParam]
 89    ) -> tuple[str, Usage | None]:
 90        """Return the SU's next message and the driver model's usage for the call.
 91
 92        `conversation` is in the eval frame and must end on an `assistant`
 93        (target) turn. Drive-loop termination is the caller's concern.
 94
 95        The whole `Usage` rather than just its cost: the SU's TaskRun is never
 96        persisted, so this in-memory value is the only place the driver model's
 97        tokens ever exist. A cost alone can neither be split per model against an
 98        invoice nor recomputed at a different price, and `cost / total_tokens`
 99        over a figure whose tokens are the agent's is meaningless.
100
101        None when the provider reported nothing — distinct from a zeroed Usage,
102        which would read as a genuinely free call rather than an unmeasured one.
103        """
104        # 1) Keep user and assistant turns (drop system/tool if present).
105        visible = [m for m in conversation if m["role"] in ("user", "assistant")]
106        # 2) Drop tool-dispatch-only assistant turns (falsy content).
107        #    See _is_tool_dispatch_only for rationale.
108        visible = [m for m in visible if not _is_tool_dispatch_only(m)]
109        # 3) Invariants.
110        if not visible:
111            raise ValueError("No LLM-visible messages in conversation.")
112        if visible[-1]["role"] != "assistant":
113            raise ValueError(
114                "Conversation must end on an assistant (target) turn — the SU "
115                "is responding to that turn."
116            )
117
118        # 4) Role-swap then assemble prior_trace + input. The last swapped
119        #    turn becomes the LLM `input`; everything before it goes into
120        #    `prior_trace` with a system message prepended.
121        swapped = role_swap(visible)
122        last = swapped[-1]
123        user_input = last["content"]
124        if not isinstance(user_input, str):
125            raise RuntimeError(
126                "synthetic user input must be a plain string after role_swap"
127            )
128
129        system_msg: ChatCompletionSystemMessageParam = {
130            "role": "system",
131            "content": self._system_prompt,
132        }
133        prior_trace: list[ChatCompletionMessageParam] = [system_msg, *swapped[:-1]]
134
135        # 5) Adapter call (in-memory, no disk write).
136        task_run, run_output = await self._adapter.invoke_returning_run_output(
137            user_input, prior_trace=prior_trace
138        )
139        raw = run_output.output
140        if not isinstance(raw, str):
141            raise RuntimeError("synthetic user returned non-string output")
142
143        return raw, task_run.usage

Plays one synthetic user across multiple turns.

Constructed once per case from the typed persona + driver config; respond(conversation) is called once per turn. Callers holding the tagged wire blob parse it first (kiln_ai.synthetic_user.parser) — blob parsing stays at the wire boundary. The adapter is built at construction time and reused across all turns of the case.

SyntheticUserDriver( synthetic_user_info: SyntheticUserInfo, driver_config: SyntheticUserDriverConfig)
53    def __init__(
54        self,
55        synthetic_user_info: SyntheticUserInfo,
56        driver_config: SyntheticUserDriverConfig,
57    ):
58        self._info = synthetic_user_info
59        self._driver_config = driver_config
60        self._system_prompt: str = render_system_prompt(self._info)
61
62        # In-memory Task; nothing is persisted. The persona-playing system
63        # prompt rides on `prior_trace[0]` each call, so this `instruction`
64        # is effectively unused at runtime — kiln_ai's MultiturnFormatter
65        # uses the first system message in `prior_trace` and skips the
66        # task's instruction when `prior_trace` is non-empty. The Task
67        # model requires a non-empty instruction, hence the placeholder.
68        self._task = Task(
69            name="synthetic_user_driver",
70            description="In-memory SU player. Not persisted.",
71            instruction=(
72                "Placeholder — the persona-playing system prompt is supplied "
73                "via prior_trace on every adapter call."
74            ),
75        )
76        # Same RunConfigProperties shape used elsewhere; `structured_output_mode`
77        # is `default` because SU output is free text (no JSON schema).
78        self._run_config = KilnAgentRunConfigProperties(
79            model_name=driver_config.model_name,
80            model_provider_name=driver_config.model_provider_name,
81            prompt_id="simple_prompt_builder",
82            structured_output_mode=StructuredOutputMode.default,
83            tools_config=ToolsRunConfig(tools=[]),
84        )
85        self._adapter = adapter_for_task(self._task, self._run_config)
async def respond( self, conversation: list[typing.Union[openai.types.chat.chat_completion_developer_message_param.ChatCompletionDeveloperMessageParam, openai.types.chat.chat_completion_system_message_param.ChatCompletionSystemMessageParam, openai.types.chat.chat_completion_user_message_param.ChatCompletionUserMessageParam, kiln_ai.utils.open_ai_types.ChatCompletionAssistantMessageParamWrapper, kiln_ai.utils.open_ai_types.ChatCompletionToolMessageParamWrapper, openai.types.chat.chat_completion_function_message_param.ChatCompletionFunctionMessageParam]]) -> tuple[str, kiln_ai.utils.usage.Usage | None]:
 87    async def respond(
 88        self, conversation: list[ChatCompletionMessageParam]
 89    ) -> tuple[str, Usage | None]:
 90        """Return the SU's next message and the driver model's usage for the call.
 91
 92        `conversation` is in the eval frame and must end on an `assistant`
 93        (target) turn. Drive-loop termination is the caller's concern.
 94
 95        The whole `Usage` rather than just its cost: the SU's TaskRun is never
 96        persisted, so this in-memory value is the only place the driver model's
 97        tokens ever exist. A cost alone can neither be split per model against an
 98        invoice nor recomputed at a different price, and `cost / total_tokens`
 99        over a figure whose tokens are the agent's is meaningless.
100
101        None when the provider reported nothing — distinct from a zeroed Usage,
102        which would read as a genuinely free call rather than an unmeasured one.
103        """
104        # 1) Keep user and assistant turns (drop system/tool if present).
105        visible = [m for m in conversation if m["role"] in ("user", "assistant")]
106        # 2) Drop tool-dispatch-only assistant turns (falsy content).
107        #    See _is_tool_dispatch_only for rationale.
108        visible = [m for m in visible if not _is_tool_dispatch_only(m)]
109        # 3) Invariants.
110        if not visible:
111            raise ValueError("No LLM-visible messages in conversation.")
112        if visible[-1]["role"] != "assistant":
113            raise ValueError(
114                "Conversation must end on an assistant (target) turn — the SU "
115                "is responding to that turn."
116            )
117
118        # 4) Role-swap then assemble prior_trace + input. The last swapped
119        #    turn becomes the LLM `input`; everything before it goes into
120        #    `prior_trace` with a system message prepended.
121        swapped = role_swap(visible)
122        last = swapped[-1]
123        user_input = last["content"]
124        if not isinstance(user_input, str):
125            raise RuntimeError(
126                "synthetic user input must be a plain string after role_swap"
127            )
128
129        system_msg: ChatCompletionSystemMessageParam = {
130            "role": "system",
131            "content": self._system_prompt,
132        }
133        prior_trace: list[ChatCompletionMessageParam] = [system_msg, *swapped[:-1]]
134
135        # 5) Adapter call (in-memory, no disk write).
136        task_run, run_output = await self._adapter.invoke_returning_run_output(
137            user_input, prior_trace=prior_trace
138        )
139        raw = run_output.output
140        if not isinstance(raw, str):
141            raise RuntimeError("synthetic user returned non-string output")
142
143        return raw, task_run.usage

Return the SU's next message and the driver model's usage for the call.

conversation is in the eval frame and must end on an assistant (target) turn. Drive-loop termination is the caller's concern.

The whole Usage rather than just its cost: the SU's TaskRun is never persisted, so this in-memory value is the only place the driver model's tokens ever exist. A cost alone can neither be split per model against an invoice nor recomputed at a different price, and cost / total_tokens over a figure whose tokens are the agent's is meaningless.

None when the provider reported nothing — distinct from a zeroed Usage, which would read as a genuinely free call rather than an unmeasured one.

class SyntheticUserDriverConfig(pydantic.main.BaseModel):
30class SyntheticUserDriverConfig(BaseModel):
31    """Per-eval runtime config for the SU's LLM driver.
32
33    No `temperature` field — runs at the chosen model's default. The driver
34    intentionally does not own temperature: the persona-playing prompt and
35    `behavior_guidance` carry style; temperature is a model-level concern
36    surfaced elsewhere when it matters.
37    """
38
39    model_name: str
40    model_provider_name: ModelProviderName

Per-eval runtime config for the SU's LLM driver.

No temperature field — runs at the chosen model's default. The driver intentionally does not own temperature: the persona-playing prompt and behavior_guidance carry style; temperature is a model-level concern surfaced elsewhere when it matters.

model_name: str
model_provider_name: kiln_ai.datamodel.datamodel_enums.ModelProviderName
model_config: ClassVar[pydantic.config.ConfigDict] = {}

Configuration for the model, should be a dictionary conforming to [ConfigDict][pydantic.config.ConfigDict].

class SyntheticUserInfo(pydantic.main.BaseModel):
550class SyntheticUserInfo(BaseModel):
551    """The synthetic user's character sheet: who they are and what they want.
552
553    This is both the persisted form on multi-turn synthetic eval inputs and
554    the runtime shape the synthetic-user driver renders its system prompt
555    from. The XML-tagged blob some wire formats carry is parsed into this at
556    the wire boundary (kiln_ai.synthetic_user.parser) — it is never stored.
557
558    extra="allow": unknown fields from newer generators survive load/save
559    round-trips instead of being dropped.
560    """
561
562    model_config = ConfigDict(extra="allow")
563
564    persona: str
565    goal: str
566    behavior_guidance: str | None = None

The synthetic user's character sheet: who they are and what they want.

This is both the persisted form on multi-turn synthetic eval inputs and the runtime shape the synthetic-user driver renders its system prompt from. The XML-tagged blob some wire formats carry is parsed into this at the wire boundary (kiln_ai.synthetic_user.parser) — it is never stored.

extra="allow": unknown fields from newer generators survive load/save round-trips instead of being dropped.

model_config = {'extra': 'allow'}

Configuration for the model, should be a dictionary conforming to [ConfigDict][pydantic.config.ConfigDict].

persona: str
goal: str
behavior_guidance: str | None
class SyntheticUserInfoParseError(builtins.ValueError):
21class SyntheticUserInfoParseError(ValueError):
22    """The blob is missing a required tag, or all required tags are empty."""

The blob is missing a required tag, or all required tags are empty.

async def drive_case_for_eval( *, seed_prompt: str, synthetic_user_info: SyntheticUserInfo, target_task: kiln_ai.datamodel.Task, target_run_config: kiln_ai.datamodel.run_config.KilnAgentRunConfigProperties, su_driver_config: SyntheticUserDriverConfig, turns: int, skills: Dict[str, kiln_ai.datamodel.Skill]) -> kiln_ai.synthetic_user.drive_loop.DriveCaseResult:
36async def drive_case_for_eval(
37    *,
38    seed_prompt: str,
39    synthetic_user_info: SyntheticUserInfo,
40    target_task: Task,
41    target_run_config: KilnAgentRunConfigProperties,
42    su_driver_config: SyntheticUserDriverConfig,
43    turns: int,
44    skills: SkillsDict,
45) -> DriveCaseResult:
46    """Drive one case in memory and return the full DriveCaseResult (never saved).
47
48    The result's leaf (`chain[-1]`) has `.trace` holding the full cumulative
49    conversation and its id is None (nothing touches disk). The result also
50    carries `su_usage` — the synthetic user model's spend, which surfaces
51    nowhere else since SU turns aren't persisted. `skills` must be preloaded
52    by the caller — the adapter raises on skill tools with no injected dict.
53    """
54    su_driver = SyntheticUserDriver(synthetic_user_info, su_driver_config)
55    adapter = adapter_for_task(
56        target_task,
57        target_run_config,
58        base_adapter_config=AdapterConfig(allow_saving=False, skills=skills),
59    )
60    input_source = DataSource(
61        type=DataSourceType.synthetic,
62        properties={
63            "model_name": su_driver_config.model_name,
64            "model_provider": su_driver_config.model_provider_name.value,
65            "adapter_name": _EVAL_DRIVER_ADAPTER_NAME,
66        },
67    )
68
69    async def _invoker(
70        *,
71        input: str,
72        prior_trace: list[ChatCompletionMessageParam] | None,
73        parent_task_run: TaskRun | None,
74    ) -> TaskRun:
75        # Chaining is deliberately dropped: parent runs are unsaved (id None)
76        # and the conversation already continues through prior_trace.
77        _ = parent_task_run
78        return await adapter.invoke(
79            input=input,
80            input_source=input_source,
81            prior_trace=prior_trace,
82        )
83
84    return await drive_case(
85        seed_prompt=seed_prompt,
86        target_invoker=_invoker,
87        su_driver=su_driver,
88        turns=turns,
89    )

Drive one case in memory and return the full DriveCaseResult (never saved).

The result's leaf (chain[-1]) has .trace holding the full cumulative conversation and its id is None (nothing touches disk). The result also carries su_usage — the synthetic user model's spend, which surfaces nowhere else since SU turns aren't persisted. skills must be preloaded by the caller — the adapter raises on skill tools with no injected dict.

def parse_synthetic_user_info(blob: str) -> SyntheticUserInfo:
33def parse_synthetic_user_info(blob: str) -> SyntheticUserInfo:
34    """Parse the tagged blob. Strict on required tags; lenient on optional."""
35    persona = _extract(blob, "persona")
36    if not persona:
37        raise SyntheticUserInfoParseError(
38            "Missing or empty required tag <persona> in synthetic_user_info blob."
39        )
40    goal = _extract(blob, "goal")
41    if not goal:
42        raise SyntheticUserInfoParseError(
43            "Missing or empty required tag <goal> in synthetic_user_info blob."
44        )
45    behavior_guidance = _extract(blob, "behavior_guidance") or None
46    return SyntheticUserInfo(
47        persona=persona,
48        goal=goal,
49        behavior_guidance=behavior_guidance,
50    )

Parse the tagged blob. Strict on required tags; lenient on optional.

def role_swap( conversation: list[typing.Union[openai.types.chat.chat_completion_developer_message_param.ChatCompletionDeveloperMessageParam, openai.types.chat.chat_completion_system_message_param.ChatCompletionSystemMessageParam, openai.types.chat.chat_completion_user_message_param.ChatCompletionUserMessageParam, kiln_ai.utils.open_ai_types.ChatCompletionAssistantMessageParamWrapper, kiln_ai.utils.open_ai_types.ChatCompletionToolMessageParamWrapper, openai.types.chat.chat_completion_function_message_param.ChatCompletionFunctionMessageParam]]) -> list[typing.Union[openai.types.chat.chat_completion_developer_message_param.ChatCompletionDeveloperMessageParam, openai.types.chat.chat_completion_system_message_param.ChatCompletionSystemMessageParam, openai.types.chat.chat_completion_user_message_param.ChatCompletionUserMessageParam, kiln_ai.utils.open_ai_types.ChatCompletionAssistantMessageParamWrapper, kiln_ai.utils.open_ai_types.ChatCompletionToolMessageParamWrapper, openai.types.chat.chat_completion_function_message_param.ChatCompletionFunctionMessageParam]]:
24def role_swap(
25    conversation: list[ChatCompletionMessageParam],
26) -> list[ChatCompletionMessageParam]:
27    """Flip eval-frame user/assistant labels into LLM-frame labels.
28
29    Only `user` and `assistant` are handled. The driver is expected to
30    have filtered out other roles before calling this.
31    """
32    result: list[ChatCompletionMessageParam] = []
33    for msg in conversation:
34        role = msg["role"]
35        if role not in ("user", "assistant"):
36            raise ValueError(
37                f"role_swap received unsupported role {role!r}; "
38                "the driver should have filtered it"
39            )
40        # The TypedDict union allows non-string content for multimodal /
41        # tool turns, but the synthetic user only ever sees plain text from
42        # the target. Narrowing here lets us assign into the swapped wrapper
43        # type without a cast.
44        content = msg["content"]
45        if not isinstance(content, str):
46            raise ValueError(
47                f"role_swap requires string content for role {role!r}; "
48                f"got {type(content).__name__}"
49            )
50        if role == "user":
51            assistant_msg: ChatCompletionAssistantMessageParamWrapper = {
52                "role": "assistant",
53                "content": content,
54            }
55            result.append(assistant_msg)
56        else:  # role == "assistant"
57            user_msg: ChatCompletionUserMessageParam = {
58                "role": "user",
59                "content": content,
60            }
61            result.append(user_msg)
62    return result

Flip eval-frame user/assistant labels into LLM-frame labels.

Only user and assistant are handled. The driver is expected to have filtered out other roles before calling this.