diff --git a/CONTEXT.md b/CONTEXT.md index edb6362bb..53d08e82f 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -1135,6 +1135,13 @@ read; whether a playbook is offered on this machine is config (the the model is shown, and `raven playbook run` still resolves it), because the file is the distribution unit and local state must not travel with it. +Schema v2 separates three concepts. A reusable artifact contains a **Harness**, a +**Workflow**, or both: the Harness is the durable worker table bound into a turn; +the Workflow is the validated DAG compiled from one accepted successful run. +A **Run Record** is the per-execution evidence and metadata kept separately from +the reusable artifact; it is neither discoverable nor loaded as a Playbook. +Schema v1 remains supported through the `dag` and `prompt` modes below. + The two modes differ in where the graph comes from, and therefore in who acts on a load: `dag` ships it as `nodes`, which the engine fills and dispatches through `SubAgentDagTool.execute` — the same entry a model-composed graph takes, so one @@ -1145,16 +1152,19 @@ model in the room, composes with one call of its own. `mode` is the author's statement of how completely they specified the procedure, and is deliberately not in the tool signature. -**Discovery is the model's, not a matcher's.** A playbook is reached through -`load_playbook`, one of the tools a turn can use, alongside `spawn` and -`run_subagent_dag` — there is no pre-turn interception and no LLM gate. -`triggers.keywords` decides which playbooks get *described* in that tool when the -library is larger than `playbooks.router.topK`; the `name` enum stays the whole -library, so a retrieval miss leaves a playbook undescribed rather than -unreachable. What the caller may supply is bounded to `params` and `fills`, and a -`fills` entry aimed at a field the playbook already wrote is refused — so a -playbook can be completed but never edited, and the file in git stays an accurate -account of what ran. +**Discovery has two paths.** Normally the main model chooses `load_playbook` +alongside `spawn` and `run_subagent_dag`. When +`playbooks.agentHarness=generate`, one pre-turn setup-model call receives the +ranked saved candidates and may select a direct match. That selection may bind +the saved durable Harness before the main turn and annotates `load_playbook`; +it never executes the stored Workflow. The main model must call +`load_playbook` to execute a Workflow. This resolver is an LLM decision, not a +passive keyword matcher and not an auto-run path. +`match.keywords` on v2 and `triggers.keywords` on v1 rank what is described +when the library exceeds `playbooks.router.topK`; the `name` enum still covers +the whole library. What the caller may supply is bounded to declared parameters +and fills, and a fill aimed at a field the Playbook already wrote is refused, so +a Playbook can be completed but never edited. _Avoid_: calling `triggers.keywords` a trigger — a keyword makes a playbook visible, never run. And avoid describing `confirm` as a playbook-level gate: it is `SubAgentDagSpec.confirm`, a graph-level parameter the playbook's value is diff --git a/i18n/messages.json b/i18n/messages.json index 28af39fef..8635c2ea1 100644 --- a/i18n/messages.json +++ b/i18n/messages.json @@ -6139,6 +6139,10 @@ "en": "composed per run", "zh": "流程每次现搭" }, + "gui.pb.shape_harness": { + "en": "Harness · {n}", + "zh": "Harness · {n}" + }, "gui.pb.more_steps": { "en": "+{n}", "zh": "+{n}" @@ -6203,6 +6207,10 @@ "en": "Graph", "zh": "图" }, + "gui.pb.tab_harness": { + "en": "Harness", + "zh": "Harness" + }, "gui.pb.tab_assembly": { "en": "Assembly guide", "zh": "编排指引" @@ -6223,6 +6231,18 @@ "en": "before dispatch", "zh": "派发前确认" }, + "gui.pb.f_artifact": { + "en": "artifact", + "zh": "产物" + }, + "gui.pb.sec_harness": { + "en": "Harness", + "zh": "Harness" + }, + "gui.pb.harness_only": { + "en": "This artifact provides reusable workers and has no stored Workflow.", + "zh": "这个产物提供可复用的 workers,不包含已保存的 Workflow。" + }, "gui.pb.sec_keywords": { "en": "Trigger keywords", "zh": "触发关键词" diff --git a/pyproject.toml b/pyproject.toml index 676904d32..7afef5912 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -191,6 +191,7 @@ include = [ "raven/providers/data/*.toml", # Playbook role-pool defaults and the builtin playbook library. "raven/playbook/*.yaml", + "raven/playbook/*.json", "raven/playbook/builtin/**/*.md", # Tracing dashboard viewer (dependency-free Node server + static client), # a CLI-launched surface, so it lives beside the CLI and not in the kernel. diff --git a/raven/agent/harness/memory.py b/raven/agent/harness/memory.py index 1404194b0..8ccf89594 100644 --- a/raven/agent/harness/memory.py +++ b/raven/agent/harness/memory.py @@ -152,9 +152,9 @@ def _briefed(turn: "TurnContext") -> "TurnContext": from raven.agent.subagent.charter import current_charter charter = current_charter() - if charter is None or not (charter.prompt or charter.stop_when): + if charter is None or not (charter.task_brief or charter.stop_when): return turn - return replace(turn, task_brief=charter.prompt, task_done_when=charter.stop_when) + return replace(turn, task_brief=charter.task_brief, task_done_when=charter.stop_when) async def shrink( self, diff --git a/raven/agent/harness_capabilities.py b/raven/agent/harness_capabilities.py new file mode 100644 index 000000000..ae03d2b56 --- /dev/null +++ b/raven/agent/harness_capabilities.py @@ -0,0 +1,95 @@ +"""Load the checked-in capability switches for generated worker harnesses.""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any, Literal + +ModuleName = Literal["memory", "planning", "capability", "action"] +FunctionKind = Literal["self", "participant"] + +_PATH = Path(__file__).parents[1] / "playbook" / "harness_generation.json" +_CONNECTED_PARAMETERS = frozenset( + { + ("memory", "systemPrompt"), + ("memory", "stopWhen"), + ("capability", "tools"), + ("action", "checks"), + } +) +_CONNECTED_FUNCTIONS = frozenset( + { + ("memory", "participant", "intake"), + ("planning", "participant", "advise"), + ("action", "participant", "judge"), + ("action", "participant", "salvage"), + } +) +_CACHE_KEY: tuple[Path, int, int] | None = None +_CACHE_ROOT: dict[str, Any] | None = None + + +def _entry_enabled(entry: Any, path: str) -> bool: + if not isinstance(entry, dict) or type(entry.get("enabled")) is not bool: + raise RuntimeError(f"invalid harness generation capability file: {path}.enabled must be a boolean") + return entry["enabled"] + + +def _validate_connected(root: dict[str, Any]) -> None: + for module, body in root.items(): + if module == "note" or not isinstance(body, dict): + continue + parameters = body.get("parameters", {}).get("items", {}) + for name, entry in parameters.items(): + path = f"{module}.parameters.{name}" + if _entry_enabled(entry, path) and (module, name) not in _CONNECTED_PARAMETERS: + raise RuntimeError(f"harness generation capability is enabled but not wired: {path}") + functions = body.get("functions", {}) + for kind in ("self", "participant"): + for name, entry in functions.get(kind, {}).get("items", {}).items(): + path = f"{module}.functions.{kind}.{name}" + if _entry_enabled(entry, path) and (module, kind, name) not in _CONNECTED_FUNCTIONS: + raise RuntimeError(f"harness generation capability is enabled but not wired: {path}") + + +def _root() -> dict[str, Any]: + global _CACHE_KEY, _CACHE_ROOT + try: + stat = _PATH.stat() + cache_key = (_PATH, stat.st_mtime_ns, stat.st_size) + if _CACHE_KEY == cache_key and _CACHE_ROOT is not None: + return _CACHE_ROOT + document = json.loads(_PATH.read_text(encoding="utf-8")) + root = document["harnessGeneration"] + except (OSError, json.JSONDecodeError, KeyError, TypeError) as exc: + raise RuntimeError(f"invalid harness generation capability file: {exc}") from exc + if not isinstance(root, dict): + raise RuntimeError("invalid harness generation capability file: harnessGeneration must be an object") + _validate_connected(root) + _CACHE_KEY = cache_key + _CACHE_ROOT = root + return root + + +def parameter_enabled(module: ModuleName, name: str) -> bool: + """Whether the generator may emit one module parameter.""" + root = _root() + try: + entry = root[module]["parameters"]["items"][name] + except (KeyError, TypeError) as exc: + raise RuntimeError(f"harness generation capability is not declared: {module}.parameters.{name}") from exc + return _entry_enabled(entry, f"{module}.parameters.{name}") + + +def function_enabled(module: ModuleName, kind: FunctionKind, name: str) -> bool: + """Whether the generator may emit one module function.""" + root = _root() + try: + entry = root[module]["functions"][kind]["items"][name] + except (KeyError, TypeError) as exc: + raise RuntimeError(f"harness generation capability is not declared: {module}.functions.{kind}.{name}") from exc + return _entry_enabled(entry, f"{module}.functions.{kind}.{name}") + + +__all__ = ["function_enabled", "parameter_enabled"] diff --git a/raven/agent/loop/main.py b/raven/agent/loop/main.py index f44886e60..75d4d3801 100644 --- a/raven/agent/loop/main.py +++ b/raven/agent/loop/main.py @@ -888,6 +888,7 @@ async def run_turn( """ session_key = req.conversation or f"{req.source.channel}:{req.source.chat_id}" flush = True + capture = None try: # Pick up a mid-session `deep-research enable` BEFORE the freeze below # captures the turn's pairs. The promotion re-registers the offer @@ -915,23 +916,56 @@ async def run_turn( # forbids: a turn resolves its pair once and holds it for the whole # turn tree. binding = self.binding_for_session(session_key) - delegate_table = await self._write_worker_table(req, session_key, binding) + resolution = await self._resolve_playbook_turn(req, session_key, binding) + delegate_table = resolution.table + if resolution.active: + from raven.playbook.run_record import PlaybookRunCapture + + capture = PlaybookRunCapture( + query=getattr(req, "text", "") or "", + disposition=resolution.disposition, + selected_playbook=resolution.selected_playbook, + artifact_name=resolution.artifact_name, + capture_workflow=resolution.capture_workflow, + ) + from raven.playbook.run_record import capture_scope + # The charter a dispatch staged for this session, taken for this turn # only. Both scopes below are None on an ordinary turn, which is the # path every reader answers to as "no playbook". charter = self._take_session_charter(session_key) + if resolution.disposition == "artifact" and resolution.artifact_name: + from raven.agent.subagent.charter import Charter + + artifact_status = ( + f"has generated and saved the reusable Persona Harness {resolution.artifact_name!r}" + if resolution.persisted + else ( + f"generated the Persona Harness {resolution.artifact_name!r} for this turn, " + "but persistence failed" + ) + ) + charter = Charter( + prompt=( + f"The platform {artifact_status}. Do not call load_playbook, search for a persona " + "format, or write Playbook, skill, persona, Harness, Workflow, or run-record files. " + "Do not spawn workers or execute a Workflow. Reply concisely with what the generated " + "Harness is for and whether it was saved." + ) + ) with ( use_binding(binding), self.tools.session_scope_for(session_key), self.tools.turn_scope(), delegate_scope(delegate_table), charter_scope(charter), + capture_scope(capture), # The participant seats in the hook chain ask this turn's modules, # so a replaced Action or Planning decides what a plugin's # judgement does -- bound per turn like the model. bind_harness(self.harness), ): - return await self._run_turn( + outcome = await self._run_turn( req, emit, drain, @@ -940,10 +974,26 @@ async def run_turn( usage_sink=usage_sink, text_sink=text_sink, ) + if capture is not None: + await self._finish_playbook_turn(resolution, capture, binding) + return outcome except asyncio.CancelledError: flush = False raise finally: + if capture is not None and capture.status == "running" and self._playbooks is not None: + from raven.playbook.run_record import RunRecordStore + + capture.finish(status="cancelled" if not flush else "failed") + try: + RunRecordStore(self._playbooks.store.root).save(capture) + except Exception: # noqa: BLE001 - cleanup cannot replace the turn's error + logger.opt(exception=True).warning("playbook: failed turn record could not be saved") + loader = self.tools.get("load_playbook") + setter = getattr(loader, "set_preselected", None) + if callable(setter): + setter(None) + # Every way a turn ends passes here, which is what makes this the # place a foreground DAG run learns its turn is over. A direct chat # runs on an instance's own lane, concurrently with the main agent's diff --git a/raven/agent/loop/turn_path.py b/raven/agent/loop/turn_path.py index e3281065e..7a523ceff 100644 --- a/raven/agent/loop/turn_path.py +++ b/raven/agent/loop/turn_path.py @@ -1788,12 +1788,40 @@ async def _process_message( # noqa: C901 (cc 47: pre-existing, above the ceilin # (content, media) reply directly. # # Skip the user-inbound hooks for Sentinel / subagent turns (by origin). + inbound_original = content skip_user_inbound = origin in _SKIP_USER_INBOUND_ORIGINS + if skip_user_inbound: + from raven.agent.harness.participants import Intake, read_intake + from raven.agent.subagent.charter import charter_participants + from raven.contracts.participant import StepView + + participants = charter_participants() + if participants: + peeked = self.sessions.peek(msg_session_key) + step = StepView( + session_key=msg_session_key, + iteration=0, + response=None, + transcript=(), + history=tuple(peeked.messages if peeked is not None else ()), + turn_base=0, + question=content, + rollbacks=0, + mode=None, + mode_overlay=None, + phase="user_inbound", + ) + answer = await self.harness.memory.ask_intake(content, step, participants) + intake = answer if answer is None or isinstance(answer, Intake) else read_intake(answer, text=content) + if intake is not None: + if intake.reply is not None: + return str(intake.reply), [] + content = intake.text + # The turn's one hook-metadata dict, and the words the user actually # sent: the first crosses every phase group with the turn, the second # is what history keeps no matter how hooks rewrite the model's view. turn_hook_meta: dict[str, Any] = {} - inbound_original = content if len(self.hooks) > 0 and not skip_user_inbound: _peeked = self.sessions.peek(msg_session_key) _hook_ctx = AgentHookContext( @@ -1805,7 +1833,10 @@ async def _process_message( # noqa: C901 (cc 47: pre-existing, above the ceilin ) _decision = await self.hooks.before_user_inbound(_hook_ctx) if _decision.short_circuit_result is not None: - return _decision.short_circuit_result + result = _decision.short_circuit_result + if isinstance(result, tuple) and len(result) == 2: + return result + return str(result), [] if _decision.modified_content is not None: content = _decision.modified_content diff --git a/raven/agent/loop/wiring.py b/raven/agent/loop/wiring.py index 788c8c276..1ba4fe47b 100644 --- a/raven/agent/loop/wiring.py +++ b/raven/agent/loop/wiring.py @@ -680,64 +680,161 @@ def _take_session_charter(self, session_key: str) -> "Charter | None": """ return self._session_charters.pop(session_key, None) - async def _write_worker_table(self, req: Any, session_key: str, binding: Any) -> "DelegateTable | None": - """This turn's worker table, or ``None`` to run it unconfigured. - - ``None`` on every path that is not a deliberate, successful generation: - the feature off, a sub-agent process (a worker writing its own workers - would be the third level the two-level rule forbids), a direct chat with - one sub-agent, an empty roster, or a generation that failed. A turn that - dies because its setup step failed is strictly worse than one that runs - without it. - - The binding is handed in rather than resolved here. It has to be the - turn's own pair, because this runs *before* ``use_binding`` opens and - ``self.provider`` still answers with the loop's default; and it has to - be resolved once for both, because this call awaits a model and a - session that switched while it was in flight would otherwise split the - turn across two pairs. - - The tool names handed over are the registry's current view, taken - outside the turn's freeze for the same reason. They are a vocabulary for - the brief, not the array the turn will run on, so a session-overlay tool - missing from them costs a word the generator could have used and - nothing else. - """ + async def _resolve_playbook_turn(self, req: Any, session_key: str, binding: Any): + """Generate, persist and bind the Harness selected by this turn's mode.""" + from raven.playbook.agent_generator import HarnessResolution + cfg = self._playbook_config if cfg is None or not cfg.enabled or getattr(cfg, "agent_harness", "default") != "generate": - return None - if is_subagent_process(): - return None - # A direct chat with one sub-agent returns through ``subagents.chat`` - # without ever rendering or executing ``spawn``, so a table written for - # it is never read. Guarded before the call rather than after: the cost - # of generating one is a model round trip (two, when the table needs a - # repair round), paid on every direct turn for nothing. - if getattr(req, "direct_target", None) is not None: - return None + return HarnessResolution() + mode = getattr(req, "playbook_mode", None) or getattr(cfg, "default_generation_mode", "task") + if mode == "off": + return HarnessResolution() + if is_subagent_process() or getattr(req, "direct_target", None) is not None: + return HarnessResolution() try: - from raven.playbook.agent_generator import WorkerTableGenerator, roster_note + from raven.playbook.agent_generator import ( + PersonaPlaybookGenerator, + TaskPlaybookGenerator, + persona_roster_profile, + task_roster_profile, + ) metas = list(self.subagents.list_agents()) agents = [a.name for a in metas] if not agents: - return None - # What each agent is for, in the registry's own words and its own - # advertised capabilities. Without them the generating model is - # handed a list of bare names and, on a roster that is not the - # shipped one, cannot tell which agent the task wants -- not even - # when only one of them can read the local files it is about. - notes = {a.name: roster_note(a) for a in metas} - tools = sorted((d.get("function", d) or {}).get("name", "") for d in self.tools.get_definitions()) - table = await WorkerTableGenerator(binding.provider, binding.model).generate( - getattr(req, "text", "") or "", agents, [t for t in tools if t], notes - ) + return HarnessResolution() + tool_catalog = self.tools.get_definitions() + query = getattr(req, "text", "") or "" + if mode == "persona": + profiles = {meta.name: persona_roster_profile(meta) for meta in metas} + names = self._playbooks.names() if self._playbooks is not None else [] + resolution = await PersonaPlaybookGenerator(binding.provider, binding.model).resolve( + query, + agents, + tool_catalog, + profiles, + names, + ) + else: + profiles = {meta.name: task_roster_profile(meta) for meta in metas} + resolution = await TaskPlaybookGenerator(binding.provider, binding.model).resolve( + query, + agents, + tool_catalog, + profiles, + ) + if resolution.active: + resolution = self._persist_generated_harness(resolution, query) except Exception: # noqa: BLE001 - setup must not cost the turn - logger.opt(exception=True).warning("agent playbook: worker table failed; running unconfigured") - return None + logger.opt(exception=True).warning("agent playbook: resolution failed; running unconfigured") + return HarnessResolution() + table = resolution.table if table: logger.info("agent playbook: {} worker(s) for this turn: {}", len(table.workers), table.labels()) - return table + return resolution + + def _persist_generated_harness(self, resolution: Any, query: str): + """Save a validated Harness now; persistence failure never blocks binding.""" + if self._playbooks is None or resolution.spec is None: + return resolution + import re + from dataclasses import replace + + from raven.playbook.unified import PlaybookMatch, UnifiedPlaybookSpec + + description = (resolution.description or query.strip().splitlines()[0])[:200] + keywords = [word.lower() for word in re.findall(r"[A-Za-z0-9][A-Za-z0-9_-]{2,}", query)[:8]] + artifact = UnifiedPlaybookSpec( + name=resolution.artifact_name or resolution.spec.name, + description=description, + match=PlaybookMatch(summary=description, keywords=keywords or [description[:80]]), + harness=resolution.spec, + ) + try: + artifact, path = self._save_generated_artifact(artifact) + self._playbooks.adopt(artifact.name) + harness = artifact.harness + logger.info("playbook: saved generated Harness {!r} at {}", artifact.name, path) + return replace( + resolution, + spec=harness, + artifact_name=artifact.name, + persisted=True, + ) + except Exception: # noqa: BLE001 - persistence must not cost the turn + logger.opt(exception=True).warning("playbook: generated Harness could not be saved; using it in memory") + return resolution + + def _save_generated_artifact(self, artifact: Any): + """Atomically save under the requested name or the first numeric suffix.""" + if self._playbooks is None: + raise RuntimeError("Playbook runtime is unavailable") + from raven.playbook.store import PlaybookExistsError + + base = artifact.name + attempt = 1 + while True: + name = base if attempt == 1 else f"{base}-{attempt}" + harness = artifact.harness + if harness is not None: + harness = harness.model_copy(update={"name": name}) + candidate = artifact.model_copy(update={"name": name, "harness": harness}) + try: + return candidate, self._playbooks.store.save(candidate) + except PlaybookExistsError: + attempt += 1 + + async def _write_worker_table(self, req: Any, session_key: str, binding: Any) -> "DelegateTable | None": + """Backward-compatible table-only face used by focused tests.""" + return (await self._resolve_playbook_turn(req, session_key, binding)).table + + async def _finish_playbook_turn(self, resolution: Any, capture: Any, binding: Any) -> None: + """Update a Task artifact with proven Workflow evidence and save the run.""" + if self._playbooks is None: + return + from raven.playbook.run_record import RunRecordStore + + try: + harness = resolution.spec + capture.saved_playbook = resolution.artifact_name if resolution.persisted else None + if resolution.capture_workflow and capture.dags and not resolution.selected_playbook: + from raven.playbook.workflow_compiler import WorkflowCompiler + + artifact = await WorkflowCompiler(binding.provider, binding.model).compile( + query=capture.query, + dag=capture.dags[-1].spec, + run_id=capture.run_id, + harness=harness, + name_hint=resolution.artifact_name, + description_hint=resolution.description, + ) + store = self._playbooks.store + if resolution.persisted: + # The compiler parameterizes a proven graph; it does not own + # artifact identity. Keep the numeric suffix chosen by the + # atomic Harness save even when the model ignores nameHint. + artifact = artifact.model_copy(update={"name": resolution.artifact_name, "harness": harness}) + path = store.save(artifact, overwrite=True) + else: + artifact, path = self._save_generated_artifact(artifact) + self._playbooks.adopt(artifact.name) + capture.saved_playbook = artifact.name + logger.info("playbook: updated Task artifact {!r} with Workflow at {}", artifact.name, path) + capture.finish() + except Exception as exc: # noqa: BLE001 - persistence must not replace the user's answer + logger.opt(exception=True).warning("playbook: turn finalization failed") + capture.finish(status="completed", error=str(exc)) + finally: + try: + path = RunRecordStore(self._playbooks.store.root).save(capture) + logger.info("playbook: saved run record {}", path) + except Exception: # noqa: BLE001 - record failure must not cost the turn + logger.opt(exception=True).warning("playbook: run record could not be saved") + loader = self.tools.get("load_playbook") + setter = getattr(loader, "set_preselected", None) + if callable(setter): + setter(None) def set_default_binding(self, binding: ModelBinding) -> None: """Change what new sessions start on. @@ -1327,10 +1424,10 @@ def _build_playbook_runtime(self, cfg: "PlaybookConfig"): # terms once judgement is wired in. provider_for=self._verdict_provider, binding_for=self._turn_binding, - # Stored Playbook nodes already name roster agents and carry their - # own prompts. A turn-scoped generated worker with the same label - # must not rewrite that persisted graph. - worker_table_for=lambda: None, + # PlaybookRuntime opens an explicit delegate scope for every stored + # graph: its durable Harness table for a v2 composite, or None for a + # legacy/workflow-only graph. Reading the scope here lets composites + # resolve aliases without exposing them to unrelated stored DAGs. control_reachable=self.dag_control_reachable, control_advert=self.dag_control_advert, verdict_config=self.subagent_dag_config, diff --git a/raven/agent/subagent/backends/raven_loop.py b/raven/agent/subagent/backends/raven_loop.py index 3e8fb5bef..cee408bf7 100644 --- a/raven/agent/subagent/backends/raven_loop.py +++ b/raven/agent/subagent/backends/raven_loop.py @@ -8,7 +8,7 @@ from __future__ import annotations import json -from collections.abc import Awaitable, Callable, Collection, Sequence +from collections.abc import Awaitable, Callable, Collection, Mapping, Sequence from pathlib import Path from typing import Any @@ -33,6 +33,7 @@ from raven.config.live import LiveConfig, exec_extra_deny_patterns from raven.config.schema import LLM_ERROR_RETRY_DELAYS_DEFAULT, ExecToolConfig from raven.contracts.llm_provider import LLMProvider +from raven.contracts.participant import StepView from raven.contracts.subagent_backend import SubagentActionAbortedError, SubagentNoAnswerError from raven.contracts.tool import SKIPPED_AFTER_BLOCKED_CALL, Continuation from raven.memory_engine import filter_by_required_tools @@ -59,6 +60,19 @@ def _live_exec_extra_deny() -> list[str] | None: return exec_extra_deny_patterns(_LIVE_CONFIG) +def _append_participant_note(messages: list[dict[str, Any]], note: str) -> None: + """Append generated advice without importing the main loop back into this backend.""" + if not messages or not note: + return + body = messages[-1].get("content") + if isinstance(body, str): + messages[-1]["content"] = f"{body}\n\n{note}" if body else note + elif isinstance(body, list): + messages[-1]["content"] = [*body, {"type": "text", "text": note}] + elif body is None: + messages[-1]["content"] = note + + def build_subagent_prompt( agent_home: Path, work_dir: Path, @@ -262,7 +276,7 @@ async def run( finally: IN_SUBAGENT_RUN.reset(token) - async def _run( + async def _run( # noqa: C901 -- this is the bounded in-process worker loop self, task: str, *, @@ -404,6 +418,43 @@ def allowed(name: str) -> bool: } ] ) + iteration = 0 + participants = charter_mod.charter_participants() + original_task = task + + async def participant_answer(verb: str, *args: Any) -> Any: + if not participants: + return None + try: + return await getattr(participants[0], verb)(*args) + except Exception: + logger.exception("generated charter participant %s raised; treating it as silence", verb) + return None + + def participant_step(phase: str, *, response: Any = None) -> StepView: + return StepView( + session_key=session_key or task_id, + iteration=iteration, + response=response, + transcript=tuple(messages), + history=tuple(history or ()), + turn_base=len(history or ()), + question=original_task, + rollbacks=0, + mode=None, + mode_overlay=None, + phase=phase, + tools=tuple(tools.get_definitions()), + max_iterations=self._MAX_ITERATIONS, + ) + + if participants: + intake = await participant_answer("intake", task, participant_step("user_inbound")) + if isinstance(intake, Mapping): + if intake.get("reply") is not None: + return str(intake["reply"]) + if isinstance(intake.get("text"), str): + task = intake["text"] messages.append({"role": "user", "content": with_attachment_note(task, media)}) # Where this run's own turns begin. Taken here rather than assumed to be # index 2, because a resumed instance arrives with its whole history in @@ -411,7 +462,6 @@ def allowed(name: str) -> bool: # node's work as this node's. own_turns_from = len(messages) - iteration = 0 final_result: str | None = None # Whether the LAST model response of this run was cut at the output # ceiling, not whether any was: a round that was cut and then answered @@ -420,6 +470,10 @@ def allowed(name: str) -> bool: cut_at_ceiling = False while iteration < self._MAX_ITERATIONS: iteration += 1 + if participants: + advice = await participant_answer("advise", participant_step("iteration")) + if isinstance(advice, str) and advice: + _append_participant_note(messages, advice) if on_delta is None: response = await provider.chat_with_retry( messages=messages, @@ -557,6 +611,10 @@ def allowed(name: str) -> bool: if getattr(result, "continuation", None) is Continuation.ABORT_TURN: raise SubagentActionAbortedError break + if participants: + advice = await participant_answer("advise", participant_step("after_iteration", response=response)) + if isinstance(advice, str) and advice: + _append_participant_note(messages, advice) else: final_result = response.content break @@ -594,6 +652,9 @@ def allowed(name: str) -> bool: # all carries the reason as well -- that is the shape this exists for. if cut_at_ceiling: activity.note_output_limit() + if final_result is None and participants: + salvaged = await participant_answer("salvage", participant_step("answerless")) + final_result = salvaged if isinstance(salvaged, str) and salvaged else None if final_result is None: # Nothing to hand back. Raised rather than returned, so the node # fails instead of completing with a sentence the next step would diff --git a/raven/agent/subagent/charter.py b/raven/agent/subagent/charter.py index 8d90075e1..40ead30e2 100644 --- a/raven/agent/subagent/charter.py +++ b/raven/agent/subagent/charter.py @@ -31,6 +31,8 @@ from loguru import logger +from raven.contracts.participant import AgentParticipant, StepView + MAX_PROMPT_CHARS = 4000 """A charter's prompt is a brief, not a second system prompt. Bounded so a malformed or hostile payload cannot push the turn's own identity out of the @@ -41,6 +43,18 @@ MAX_CODE_CHARS = 8000 +def _parameter_enabled(module: str, name: str) -> bool: + from raven.agent.harness_capabilities import parameter_enabled + + return parameter_enabled(module, name) + + +def _function_enabled(module: str, kind: str, name: str) -> bool: + from raven.agent.harness_capabilities import function_enabled + + return function_enabled(module, kind, name) + + @dataclass(frozen=True) class CheckRule: """One judgement about a call the worker is about to make.""" @@ -59,9 +73,12 @@ class Charter: """One dispatch's brief, already narrowed.""" prompt: str = "" - """Appended after this agent's own identity, never in place of it: replacing - it would drop the runtime facts (working directory, platform policy, the - untrusted-content rule) that the identity segment carries.""" + """The dispatch brief, separate from generated instructions so either can + be disabled without erasing or impersonating the other.""" + + instruction_addendum: str = "" + """Task-specific instructions appended after the worker's existing identity, + never in place of the runtime facts and safety rules that identity carries.""" tools: tuple[str, ...] | None = None """``None`` is "whatever this agent already offers". A tuple narrows to @@ -76,6 +93,9 @@ class Charter: source that does not pass never runs, and one that raises at run time has said nothing rather than refused everything.""" + functions: tuple[tuple[str, str], ...] = () + """Enabled generated participant sources, stored as immutable name-source pairs.""" + timeout_s: int | None = None """A deadline for this dispatch, if the brief carried one. Tightening only: a worker configured with a limit keeps the smaller of the two, and one @@ -84,9 +104,21 @@ class Charter: stop_when: str = "" + @property + def task_brief(self) -> str: + """The brief and enabled instruction delta rendered for Memory.""" + return "\n\n".join(part for part in (self.prompt, self.instruction_addendum) if part) + def __bool__(self) -> bool: return bool( - self.prompt or self.tools is not None or self.checks or self.code or self.timeout_s or self.stop_when + self.prompt + or self.instruction_addendum + or self.tools is not None + or self.checks + or self.code + or self.functions + or self.timeout_s + or self.stop_when ) @@ -101,10 +133,21 @@ def parse(payload: Any) -> Charter | None: """ if not isinstance(payload, dict): return None - prompt = str(payload.get("prompt") or "")[:MAX_PROMPT_CHARS] - stop_when = str(payload.get("stopWhen") or payload.get("stop_when") or "")[:MAX_PROMPT_CHARS] + legacy_prompt = payload.get("prompt") + raw_brief = payload.get("brief") if "brief" in payload else legacy_prompt + prompt = str(raw_brief or "")[:MAX_PROMPT_CHARS] + instruction_addendum = ( + str(payload.get("instructionAddendum") or "")[:MAX_PROMPT_CHARS] + if _parameter_enabled("memory", "systemPrompt") + else "" + ) + stop_when = ( + str(payload.get("stopWhen") or payload.get("stop_when") or "")[:MAX_PROMPT_CHARS] + if _parameter_enabled("memory", "stopWhen") + else "" + ) tools: tuple[str, ...] | None = None - raw_tools = payload.get("tools") + raw_tools = payload.get("tools") if _parameter_enabled("capability", "tools") else None if isinstance(raw_tools, list): tools = tuple(str(name) for name in raw_tools[:MAX_TOOLS] if isinstance(name, str) and name) checks: list[CheckRule] = [] @@ -112,7 +155,7 @@ def parse(payload: Any) -> Charter | None: # from another process, and a non-list here (a bare number, an object the # sender meant as a single rule) is a ``TypeError`` raised inside the # request handler rather than the log line this function promises. - raw_checks = payload.get("checks") + raw_checks = payload.get("checks") if _parameter_enabled("action", "checks") else None for raw in raw_checks[:MAX_CHECKS] if isinstance(raw_checks, list) else (): if not isinstance(raw, dict) or not raw.get("tool"): continue @@ -128,15 +171,29 @@ def parse(payload: Any) -> Charter | None: message=str(raw.get("message") or ""), ) ) + function_modules = {"intake": "memory", "advise": "planning", "salvage": "action"} + raw_functions = payload.get("functions") + functions: list[tuple[str, str]] = [] + if isinstance(raw_functions, dict): + for name, source in raw_functions.items(): + module = function_modules.get(name) + if module and isinstance(source, str) and source.strip() and _function_enabled(module, "participant", name): + functions.append((name, source[:MAX_CODE_CHARS])) raw_timeout = payload.get("timeoutSeconds") or payload.get("timeout_s") timeout_s: int | None = None if isinstance(raw_timeout, (int, float)) and raw_timeout > 0: timeout_s = int(raw_timeout) charter = Charter( prompt=prompt, + instruction_addendum=instruction_addendum, tools=tools, checks=tuple(checks), - code=str(payload.get("code") or "")[:MAX_CODE_CHARS], + code=( + str(payload.get("code") or "")[:MAX_CODE_CHARS] + if _function_enabled("action", "participant", "judge") + else "" + ), + functions=tuple(functions), timeout_s=timeout_s, stop_when=stop_when, ) @@ -239,7 +296,46 @@ def _prior_satisfied( return False -class CharterParticipant: +def _pure_data(value: Any) -> Any: + """Detach the JSON-like part of a StepView from host runtime objects.""" + if value is None or isinstance(value, (str, int, float, bool)): + return value + if isinstance(value, Mapping): + return {str(key): _pure_data(item) for key, item in value.items()} + if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)): + return [_pure_data(item) for item in value] + return str(value) + + +def _step_payload(step: Any) -> dict[str, Any]: + response = getattr(step, "response", None) + response_data = None + if response is not None: + response_data = { + name: _pure_data(getattr(response, name, None)) + for name in ("content", "reasoning", "finish_reason", "tool_calls", "usage") + if getattr(response, name, None) is not None + } + return { + "session_key": getattr(step, "session_key", ""), + "iteration": getattr(step, "iteration", 0), + "response": response_data, + "transcript": _pure_data(getattr(step, "transcript", ())), + "history": _pure_data(getattr(step, "history", ())), + "turn_base": getattr(step, "turn_base", 0), + "question": getattr(step, "question", ""), + "rollbacks": getattr(step, "rollbacks", 0), + "mode": getattr(step, "mode", None), + "mode_overlay": _pure_data(getattr(step, "mode_overlay", None)), + "phase": getattr(step, "phase", ""), + "tools": _pure_data(getattr(step, "tools", ())), + "window": getattr(step, "window", None), + "max_iterations": getattr(step, "max_iterations", None), + "tools_ran": bool(getattr(step, "tools_ran", False)), + } + + +class CharterParticipant(AgentParticipant): """This dispatch's own judgements, as a participant. The thin shell the design calls for: a Charter is data the model sent, and @@ -247,12 +343,21 @@ class CharterParticipant: being read by the role itself. Stateless, so one instance serves every call of a turn; what it judges comes from the scope, not from this object. - Only ``judge`` today. The other verbs stay at their defaults, which is what - "a Charter carries no opinion about that" already means. + Generated pure-data functions implement only their declared verbs; every + other verb keeps the base participant's no-op answer. """ __slots__ = () + async def intake(self, text: str, step: StepView) -> Any: + return _generated_answer("intake", text, _step_payload(step)) + + async def advise(self, step: StepView) -> Any: + return _generated_answer("advise", _step_payload(step)) + + async def salvage(self, step: StepView) -> Any: + return _generated_answer("salvage", _step_payload(step)) + def judge( self, name: str, @@ -266,7 +371,7 @@ def charter_participants() -> tuple[CharterParticipant, ...]: """The participants this dispatch brings, or none when it brought no judgements. Asked after the plugins' own, so a product's rules speak first.""" charter = current_charter() - if charter is None or not (charter.checks or charter.code): + if charter is None or not (charter.checks or charter.code or charter.functions): return () return (CharterParticipant(),) @@ -305,17 +410,42 @@ def judge( return refusals -_COMPILED: dict[str, Any] = {} +_COMPILED: dict[Any, Any] = {} MAX_COMPILED = 64 -"""How many distinct judges one process keeps compiled. +"""How many distinct generated functions one process keeps compiled. -Keyed by source, and a worker process outlives any one dispatch, so without a +Generated participants are keyed by verb and source; judges by source. A worker process outlives any one dispatch, so without a ceiling this grows with every brief the process is ever sent. Cleared rather than evicted one at a time: the cost of a miss is one parse-tree walk, and a policy that has to decide *which* entry to drop is more machinery than the thing it manages.""" +def _generated_answer(name: str, *args: Any) -> Any: + charter = current_charter() + source = dict(charter.functions).get(name) if charter is not None else None + if not source: + return None + key = (name, source) + compiled = _COMPILED.get(key) + if compiled is None: + try: + from raven.agent.subagent.charter_code import compile_function + + compiled = compile_function(source, name) + except Exception as exc: # noqa: BLE001 - refused generated code is silence + logger.warning("charter: its {} function was refused ({}); ignoring it", name, exc) + compiled = False + if len(_COMPILED) >= MAX_COMPILED: + _COMPILED.clear() + _COMPILED[key] = compiled + if compiled is False: + return None + from raven.agent.subagent.charter_code import run_function + + return run_function(compiled, *args) + + def _code_refusals( charter: Charter, name: str, diff --git a/raven/agent/subagent/charter_code.py b/raven/agent/subagent/charter_code.py index 63914bfaa..3400f963a 100644 --- a/raven/agent/subagent/charter_code.py +++ b/raven/agent/subagent/charter_code.py @@ -28,6 +28,7 @@ import ast import builtins from collections.abc import Mapping, Sequence +from copy import deepcopy from typing import Any from loguru import logger @@ -50,6 +51,8 @@ ast.Compare, ast.BoolOp, ast.UnaryOp, + ast.UAdd, + ast.USub, ast.BinOp, ast.Add, ast.Sub, @@ -94,7 +97,24 @@ """ ALLOWED_CALLS: frozenset[str] = frozenset( - {"len", "str", "int", "bool", "any", "all", "sorted", "set", "list", "dict", "tuple", "min", "max", "abs"} + { + "len", + "str", + "int", + "float", + "bool", + "isinstance", + "any", + "all", + "sorted", + "set", + "list", + "dict", + "tuple", + "min", + "max", + "abs", + } ) ALLOWED_METHODS: frozenset[str] = frozenset( @@ -102,8 +122,13 @@ ) ENTRY = "judge" -"""The function a charter's code must define: ``judge(name, params, prior)``, -answering with the refusal sentences for this call.""" +SIGNATURES: dict[str, tuple[str, ...]] = { + "intake": ("text", "step"), + "advise": ("step",), + "salvage": ("step",), + "judge": ("name", "params", "prior"), +} +"""Generated functions and the exact pure-data arguments the loop supplies.""" class CharterCodeError(Exception): @@ -143,12 +168,11 @@ def _safe_builtins() -> dict[str, Any]: return {name: getattr(builtins, name) for name in ALLOWED_CALLS} -def compile_judge(source: str) -> Any: - """Compile a charter's judge, or raise :class:`CharterCodeError`. - - The gate runs on the parse tree, before anything is compiled, so a refused - source was never executable rather than executable-but-unlucky. - """ +def compile_function(source: str, entry: str) -> Any: + """Compile one generated participant function behind the AST gate.""" + expected = SIGNATURES.get(entry) + if expected is None: + raise CharterCodeError(f"unsupported charter function {entry!r}") if not source.strip(): raise CharterCodeError("empty source") if len(source) > MAX_SOURCE_CHARS: @@ -158,14 +182,41 @@ def compile_judge(source: str) -> Any: except SyntaxError as exc: raise CharterCodeError(f"syntax error: {exc.msg}") from exc if any(not isinstance(node, ast.FunctionDef) for node in tree.body): - raise CharterCodeError("a charter judge is function definitions and nothing else") + raise CharterCodeError("charter source is function definitions and nothing else") _check_tree(tree) - if ENTRY not in [node.name for node in tree.body if isinstance(node, ast.FunctionDef)]: - raise CharterCodeError(f"must define {ENTRY}(name, params, prior)") + definitions = {node.name: node for node in tree.body if isinstance(node, ast.FunctionDef)} + target = definitions.get(entry) + if target is None: + raise CharterCodeError(f"must define {entry}({', '.join(expected)})") + args = target.args + actual = tuple(arg.arg for arg in args.args) + if ( + len(actual) != len(expected) + or args.posonlyargs + or args.vararg + or args.kwarg + or args.kwonlyargs + or args.defaults + ): + raise CharterCodeError(f"{entry} must have signature {entry}({', '.join(expected)})") namespace: dict[str, Any] = {"__builtins__": _safe_builtins()} code = compile(tree, "", "exec") exec(code, namespace) # noqa: S102 - the gate above is the boundary, not this line - return namespace[ENTRY] + return namespace[entry] + + +def compile_judge(source: str) -> Any: + """Backward-compatible compiler for ``judge(name, params, prior)``.""" + return compile_function(source, ENTRY) + + +def run_function(function: Any, *args: Any) -> Any: + """Run with detached pure-data inputs; a failure is participant silence.""" + try: + return function(*(deepcopy(arg) for arg in args)) + except Exception as exc: # noqa: BLE001 - generated code must not cost the turn + logger.warning("charter function raised ({}); treating its answer as silence", exc) + return None def run_judge( @@ -203,6 +254,8 @@ def run_judge( "ENTRY", "MAX_SOURCE_CHARS", "CharterCodeError", + "compile_function", "compile_judge", + "run_function", "run_judge", ] diff --git a/raven/agent/subagent/dag_tool.py b/raven/agent/subagent/dag_tool.py index ac427cfde..76403f2ae 100644 --- a/raven/agent/subagent/dag_tool.py +++ b/raven/agent/subagent/dag_tool.py @@ -41,7 +41,7 @@ from collections.abc import Awaitable, Callable, Mapping from contextvars import ContextVar from copy import deepcopy -from dataclasses import dataclass +from dataclasses import dataclass, replace from pathlib import Path from typing import TYPE_CHECKING, Any @@ -125,6 +125,29 @@ def _with_notices(result: "str | ToolResult", notices: list[str]) -> "str | Tool return f"{head}\n\n{result}" +def _with_capture_notice(result: str | ToolResult) -> str | ToolResult: + notice = ( + "Playbook capture is automatic for this turn. The platform will compile and save this " + "graph only after a clean success. Do not write, reconstruct, or separately save " + "Playbook, Harness, Workflow, or run-record files." + ) + if isinstance(result, ToolResult): + return replace(result, model_text=f"{result.model_text}\n\n{notice}") + return f"{result}\n\n{notice}" + + +def _successful_final(event: Any) -> bool: + """Whether a foreground graph reached a clean terminal result.""" + if not isinstance(event, Final) or event.stopped or not isinstance(event.result, DagRunResult): + return False + summary = event.result.summary or {} + return ( + bool(summary.get("total")) + and summary.get("completed") == summary.get("total") + and not any(summary.get(key, 0) for key in ("failed", "cancelled", "skipped")) + ) + + @dataclass(frozen=True) class _DagOrigin: """Per-turn reply address for a run's progress and its announce. @@ -927,6 +950,16 @@ def description(self) -> str: "when a single `spawn` is the better choice. Do not design the graph from this " "description and the node schema alone. " ) + capture_note = "" + from raven.playbook.run_record import workflow_capture_requested + + if workflow_capture_requested(): + capture_note = ( + " Playbook capture is active for this turn: put the complete reusable process in " + "one graph. After a clean success the platform automatically compiles and saves " + "the Harness, Workflow, and run record. Do not inspect the run to reconstruct it, " + "and do not write or separately save Playbook files." + ) # When to reach for a DAG at all is the always-injected guide's job, not # this description's: the model reads the digest before it picks a tool, # and two resident surfaces stating the trigger differently is how they @@ -937,7 +970,7 @@ def description(self) -> str: "dependents through files (large outputs never enter your context). One call carries the " "whole graph -- do not issue a separate call per node. The graph runs in the background and " "its result is announced to you when it finishes, so do not poll it and do not re-submit it. " - f"{guide}" + f"{guide}{capture_note}" f"Available sub-agents for the `subagent` field: {names}." ) @@ -1137,13 +1170,26 @@ async def execute( "Error: run_subagent_dag is not available inside a sub-agent run — " "only the main agent orchestrates DAGs. Complete the assigned task directly." ) + from raven.playbook.run_record import current_capture + + playbook_capture = current_capture() + capture_workflow = bool(playbook_capture is not None and playbook_capture.capture_workflow) + if capture_workflow: + background = False # Wraps the whole call, not just the grant loop: a backgrounded run keeps # the context this task held when ``create_task`` copied it, so an # in-process node re-resolving its grant mid-run still finds the run's own # definitions. Reset on the way out, so the turn that dispatched a # background run does not carry them into whatever it does next. with run_mcp_scope(mcp_servers, scope=mcp_scope, credential_gaps=mcp_credential_gaps): - return await self._execute(nodes, background, confirm=confirm, task_summary=task_summary) + return await self._execute( + nodes, + background, + confirm=confirm, + task_summary=task_summary, + capture_workflow=capture_workflow, + playbook_capture=playbook_capture, + ) async def _execute( self, @@ -1151,6 +1197,8 @@ async def _execute( background: bool, confirm: bool = False, task_summary: str = "", + capture_workflow: bool = False, + playbook_capture: Any = None, ) -> str | ToolResult: # Refused whole rather than per node, and ahead of validation, for the # same reason validation runs early: a refused graph must cost zero @@ -1166,7 +1214,8 @@ async def _execute( # enough for the model to just fix and re-submit. Backgrounding must not # turn a malformed graph into an announcement that arrives a turn later. try: - spec = parse_dag_spec({"task_summary": task_summary, "nodes": nodes, "confirm": confirm}) + submitted = parse_dag_spec({"task_summary": task_summary, "nodes": nodes, "confirm": confirm}) + spec = submitted validate_and_order(spec, self._reference_roots(), await self._session_nodes()) pre = await self._preflight(spec) spec, dispatch_backends, notices, capabilities = pre.spec, pre.backends, pre.notices, pre.capabilities @@ -1198,10 +1247,19 @@ async def _execute( # After validation so a rejected graph mints nothing, and before either # mode starts so the foreground and background paths share one site. spec, auto_instances = self._mint_missing_instances(spec, capabilities) - return _with_notices( - await self._dispatch(spec, run_id, dirs, dispatch_backends, auto_instances, origin, call_id, background), - notices, + result = await self._dispatch( + spec, + run_id, + dirs, + dispatch_backends, + auto_instances, + origin, + call_id, + background, + capture=(playbook_capture, submitted) if capture_workflow else None, ) + result = _with_notices(result, notices) + return _with_capture_notice(result) if capture_workflow else result async def _dispatch( self, @@ -1213,6 +1271,7 @@ async def _dispatch( origin: _DagOrigin, call_id: str | None, background: bool, + capture: tuple[Any, SubAgentDagSpec] | None = None, ) -> str | ToolResult: """Start a validated, minted spec running and return its first result. @@ -1235,7 +1294,7 @@ async def _dispatch( self._outboxes[run_id] = outbox task = asyncio.create_task( - self._run_detached(spec, run_id, cancel, origin, dirs, call_id, auto_instances, backends, outbox) + self._run_detached(spec, run_id, cancel, origin, dirs, call_id, auto_instances, backends, outbox, capture) ) self._runs[run_id] = task @@ -1657,6 +1716,7 @@ async def _run_detached( auto_instances: frozenset[str], dispatch_backends: dict[str, Any], outbox: Outbox | None, + capture: tuple[Any, SubAgentDagSpec] | None = None, ) -> None: """Run a graph as its own task, then hand the result on. @@ -1666,7 +1726,16 @@ async def _run_detached( """ try: result = await self._run( - spec, run_id, cancel, origin, dirs, call_id, auto_instances, dispatch_backends, outbox=outbox + spec, + run_id, + cancel, + origin, + dirs, + call_id, + auto_instances, + dispatch_backends, + outbox=outbox, + capture=capture, ) except asyncio.CancelledError: if outbox is not None: @@ -1824,6 +1893,7 @@ async def _run( auto_instances: frozenset[str], dispatch_backends: dict[str, Any], outbox: Outbox | None = None, + capture: tuple[Any, SubAgentDagSpec] | None = None, ) -> str | ToolResult | DagRunResult: """Execute one validated graph and render its outcome. @@ -1918,6 +1988,11 @@ async def _to_outbox( self._desks.pop(run_id, None) # A terminal event carrying the authoritative manifest, so the web UI can + if capture is not None and _successful_final(Final(result, stopped=cancel.is_set())): + from raven.playbook.run_record import record_completed_dag + + record_completed_dag(capture[1], run_id, capture[0]) + # rebuild / finalize the graph (and survive a reload). await emit( "dag_run_completed", diff --git a/raven/agent/subagent/delegate.py b/raven/agent/subagent/delegate.py index 8d5810707..e304f797d 100644 --- a/raven/agent/subagent/delegate.py +++ b/raven/agent/subagent/delegate.py @@ -108,9 +108,20 @@ def current_delegate() -> DelegateTable | None: return _TABLE.get() +def bind_delegate_for_turn(table: DelegateTable | None) -> None: + """Replace the table inside the caller's existing turn scope. + + ``load_playbook`` runs after the outer scope has opened; setting the same + ContextVar here makes a loaded durable Harness visible for the remainder of + that turn, and the outer scope's token still restores the previous value. + """ + _TABLE.set(table or None) + + __all__ = [ "DelegateTable", "Worker", + "bind_delegate_for_turn", "current_delegate", "delegate_scope", "dispatch_charter", diff --git a/raven/agent/tools/create_playbook.py b/raven/agent/tools/create_playbook.py index c4116ceb8..eda7178de 100644 --- a/raven/agent/tools/create_playbook.py +++ b/raven/agent/tools/create_playbook.py @@ -127,13 +127,15 @@ async def execute(self, name: str, workflow: str, skills: list[str] | None = Non "Pick another name, or ask to revise the existing one." ) try: - generated = await self._generator.generate(workflow, skills) + generated = await self._generator.generate(workflow, skills, dag_only=True) except PlaybookGenerationError as exc: return ( f"Error: playbook generation failed: {exc}. Do not write directly into the Playbook " "library or report success; a Playbook is usable only after its official validation passes." ) - spec = generated.spec.model_copy(update={"name": name}) + from raven.playbook.unified import unified_from_legacy + + spec = unified_from_legacy(generated.spec, name=name) try: path = self._store.save(spec, notes=generated.notes) except PlaybookExistsError: diff --git a/raven/agent/tools/load_playbook.py b/raven/agent/tools/load_playbook.py index 0d8cb21f7..691666622 100644 --- a/raven/agent/tools/load_playbook.py +++ b/raven/agent/tools/load_playbook.py @@ -7,13 +7,13 @@ the party that has the context. **Loading is one action; what it leads to is the playbook's business.** A ``dag`` -playbook dispatches from inside this call -- the caller never gets a chance to -"load and then not run", and never sees the graph it would otherwise be tempted to -edit. A ``prompt`` playbook comes back as composition guidance for the caller to -build a graph from. Which of those happens is the author's ``mode``, and it is -deliberately absent from this tool's signature: it describes how thoroughly the -author specified their procedure, which is not something the caller should have to -classify correctly before it can ask. +Workflow dispatches from inside this call -- the caller never gets a chance to +"load and then not run", and never sees the graph it would otherwise be tempted +to edit. A Harness-only artifact activates its reusable workers for the rest of +the turn. A legacy ``prompt`` playbook still comes back as composition guidance. +Which of those happens is deliberately absent from this tool's signature: it is +the stored artifact's concern, not something the caller should classify before +it can ask. What the caller may supply is bounded to two arguments -- ``params`` (declared values) and ``fills`` (fields the author left blank) -- and ``fills`` is refused @@ -24,6 +24,7 @@ from __future__ import annotations +from contextvars import ContextVar from typing import TYPE_CHECKING, Any from raven.contracts.tool import Tool @@ -49,6 +50,7 @@ def __init__(self, runtime: "PlaybookRuntime | None" = None) -> None: self._turn_message = "" self._turn_view: tuple[list[tuple[str, str]], list[str]] | None = None self._view_revision = -1 + self._preselected: ContextVar[str | None] = ContextVar(f"load_playbook_preselected_{id(self)}", default=None) def bind_runtime(self, handles: "RuntimeHandles") -> None: """Receive the loop's assembled funnel; decline when there is none. @@ -77,6 +79,10 @@ def set_turn_message(self, message: str) -> None: self._turn_message = message or "" self._turn_view = None + def set_preselected(self, name: str | None) -> None: + """Expose the pre-turn resolver's choice to the model for this turn.""" + self._preselected.set(name) + def _library_view(self) -> tuple[list[tuple[str, str]], list[str]]: if self._runtime is None: # Admission reads ``description`` and ``parameters`` when the tool @@ -110,21 +116,29 @@ def description(self) -> str: described = {pid for pid, _ in listing} rest = [n for n in names if n not in described] more = f"\nAlso installed (ask by name for details): {', '.join(rest)}." if rest else "" + preselected = self._preselected.get() + selected = ( + f"\nThe pre-turn resolver selected {preselected!r} for this request. " + "Call load_playbook with that exact name before improvising." + if preselected in names + else "" + ) return ( - "Use one of the user's stored playbooks -- a saved multi-step procedure with its own " - "agents, parameters and steps. Reach for one when the request is the thing a playbook " - "already describes; a playbook encodes how the user wants this kind of work done, so it " - "beats improvising the same steps. If none fits, do not force it: use `spawn` or " - "`run_subagent_dag`, or just do the work.\n" - "Loading runs it. Most playbooks ship their whole graph, so one call is the whole " - "interaction; some instead come back with guidance for you to build the graph from, and " - "some ask for values first. Pass every parameter you can read off the conversation -- " + "Use one of the user's stored playbooks -- a reusable artifact that may provide a " + "Harness of task-specific workers, a validated Workflow, or both. Reach for one when " + "the request is the thing a playbook already describes; it encodes how the user wants " + "this kind of work done, so it beats improvising the same setup or steps. If none fits, " + "do not force it: use `spawn` or `run_subagent_dag`, or just do the work.\n" + "Loading activates its Harness and runs its Workflow when it has one. A Harness-only " + "artifact returns with those workers ready for this turn. Legacy prompt-mode playbooks " + "instead return guidance for you to build a graph. Some artifacts ask for values first. " + "Pass every parameter you can read off the conversation -- " "invent nothing -- and `fills` for any field listed below as left for you. A secret " "param marked as stored on this machine is filled in at load: call without it, and never " "ask the user to type a secret into the conversation. One marked as not set does not stop " "the run either -- load it anyway; the servers that param fills run without it and the " "reply says where the user sets it.\n" - f"Installed playbooks:\n{lines}{more}" + f"Installed playbooks:\n{lines}{more}{selected}" ) @property diff --git a/raven/cli/playbook_commands.py b/raven/cli/playbook_commands.py index 600e1676c..16451040d 100644 --- a/raven/cli/playbook_commands.py +++ b/raven/cli/playbook_commands.py @@ -103,7 +103,7 @@ def playbook_validate( """Validate without running: spec shape plus the field definition's rules.""" from pydantic import ValidationError - from raven.playbook.validate import validate_structure + from raven.playbook.runtime import validation_errors config = _load_config() as_path = Path(target) @@ -145,7 +145,7 @@ def playbook_validate( registry = AgentRegistry() registry.apply(config.subagents.agents) - errors.extend(validate_structure(spec, known_agents=registry.all_names())) + errors.extend(validation_errors(spec, list(registry.all_names()))) if errors: text = path.read_text(encoding="utf-8") for error in errors: @@ -214,8 +214,10 @@ def playbook_create( live_inventory(config.tools.mcp_servers), model=config.playbooks.model, ) - generated = asyncio.run(generator.generate("\n\n".join(pieces))) - spec = generated.spec.model_copy(update={"name": name}) + generated = asyncio.run(generator.generate("\n\n".join(pieces), dag_only=True)) + from raven.playbook.unified import unified_from_legacy + + spec = unified_from_legacy(generated.spec, name=name) path = store.save(spec, notes=generated.notes) console.print(f"[green]Created[/green] {escape(str(path))}") if generated.notes: diff --git a/raven/config/schema.py b/raven/config/schema.py index f04d6ee32..63fcce61e 100644 --- a/raven/config/schema.py +++ b/raven/config/schema.py @@ -2189,10 +2189,10 @@ class PlaybookConfig(Base): """Whether each turn writes itself a worker table before it starts. ``default`` is the flow this repo has always run: nothing is generated and - no new code is on the request path. ``generate`` spends one model call per - turn deciding which sub-agents the question needs and what each one's brief - is, then offers those workers -- rather than the bare roster -- to the - dispatching model. + no new code is on the request path. ``generate`` enables the pre-turn mode + chosen by ``defaultGenerationMode`` or the request override: Task selects + existing workers with minimal prompts, while Persona may build the full + enabled Harness surface. What it does not do is configure the main agent: it keeps every tool it had and decides for itself who to hand work to. The brief travels as a preamble @@ -2201,6 +2201,13 @@ class PlaybookConfig(Base): in Raven, and the enforcement point is ``ToolRegistry.execute``. """ + default_generation_mode: Literal["off", "task", "persona"] = "task" + """Default pre-turn Playbook generation mode when a request does not + override it. The off value preserves the ordinary turn path, task selects + and configures existing workers for a reusable task, and persona generates + a full durable Harness. The request-level value is a UI/session choice; + this field is the deployment fallback.""" + class SubagentsConfig(Base): """The one table of agents raven can dispatch to. diff --git a/raven/core/runtime.py b/raven/core/runtime.py index 6eac8adaf..bfd5db653 100644 --- a/raven/core/runtime.py +++ b/raven/core/runtime.py @@ -197,13 +197,18 @@ def build_runtime( memory=MemoryStore(config.workspace_path), config=ec_config.eval_engine, ) - if eval_engine is not None or plugin_hooks: - host = replace( - host, - hooks=hooks_stack.build_hooks_stack( - eval_engine=eval_engine, plugin_hooks=plugin_hooks, extra_hooks=host.hooks - ), - ) + from raven.agent.hook.participant import ParticipantHook + from raven.agent.subagent.charter import CharterParticipant + + charter_hook = ParticipantHook("generated-charter", CharterParticipant, rolls_back=False) + host = replace( + host, + hooks=hooks_stack.build_hooks_stack( + eval_engine=eval_engine, + plugin_hooks=plugin_hooks, + extra_hooks=[charter_hook, *(host.hooks or ())], + ), + ) loop = agent_loop.AgentLoop( provider=provider, workspace=config.workspace_path, diff --git a/raven/playbook/__init__.py b/raven/playbook/__init__.py index ff9f6fd81..2f467cfb2 100644 --- a/raven/playbook/__init__.py +++ b/raven/playbook/__init__.py @@ -1,17 +1,17 @@ -"""Playbook — reusable task templates: generation, matching, execution. +"""Playbook — reusable Harnesses and Workflows. -Generation, discovery by the model, execution. A playbook is one directory -under the playbooks scan root holding a -``playbook.md`` (two-field frontmatter, human body, one fenced -``yaml playbook-spec`` block) — one file, no sidecar. The two modes differ -only in where the graph comes from: ``dag`` ships it, ``prompt`` ships -assembly guidance a model turns into a graph at run time — same validation, -same execution chain. Whether a playbook is offered on this machine is config +A playbook is one directory under the scan root holding one ``playbook.md``: +two-field frontmatter, a human-readable body, and one fenced +``yaml playbook-spec`` block. Schema v2 stores an optional durable Harness, +an optional concrete DAG Workflow, or both. New artifacts never use prompt +mode. Schema-v1 DAG/prompt files remain readable and executable for backwards +compatibility. Whether an artifact is offered on this machine is config (``playbooks.disabled``), never file content. Package layout: -- ``types`` — the pydantic contract (:class:`PlaybookSpec` et al.) +- ``unified`` — schema-v2 Harness + Workflow contract +- ``types`` — legacy schema-v1 contract - ``agent_profiles`` — what each sub-agent on the table can be asked to do - ``llm_result`` — the shapes a generation call comes back in - ``params`` — parameter values and ``${params.x}`` / ``{{ params.x }}`` refs @@ -50,12 +50,21 @@ PlaybookSpec, Triggers, ) +from raven.playbook.unified import ( + InputSchema, + PlaybookMatch, + PlaybookMetadata, + StoredPlaybook, + UnifiedPlaybookSpec, + WorkflowSpec, +) __all__ = [ "BUILTIN_ROOT", "CapabilityInventory", "ExecutionPlan", "GeneratedPlaybook", + "InputSchema", "NodeSpec", "ParamSpec", "PlaybookExecutor", @@ -65,6 +74,9 @@ "PlaybookProviderError", "PlaybookProtocolError", "PlaybookOrigin", + "PlaybookMatch", + "PlaybookMetadata", + "StoredPlaybook", "MAX_GAP_ROUNDS", "PlaybookRuntime", "RouterSizes", @@ -74,6 +86,8 @@ "StaticInventory", "TriggerIndex", "Triggers", + "UnifiedPlaybookSpec", + "WorkflowSpec", "agent_profiles_from_registry", "find_collisions", "live_inventory", diff --git a/raven/playbook/agent_generator.py b/raven/playbook/agent_generator.py index da2f90c6a..3a578b9c5 100644 --- a/raven/playbook/agent_generator.py +++ b/raven/playbook/agent_generator.py @@ -1,28 +1,22 @@ -"""Write the worker table for the question that just arrived. - -What the model is asked for is deliberately small. With no graph there are no -nodes to name, no edges to keep acyclic and no per-node agent assignment -- -which is where a graph-shaped generator spends both its tokens and its failure -modes. What is left is: which of these agents does this question need, and what -is each one's brief. - -The candidate lists are handed in rather than described. A roster the host -renders into the tool's own ``enum`` cannot be hallucinated, so "an agent that -does not exist" stops being a repair round and becomes a shape the request -could not express. Only the judgements that remain -- which agents, how many, -what brief -- can be wrong, and those are what a repair round can fix. - -One repair round, then the turn runs unconfigured. A turn that dies because its -configuration step failed is strictly worse than a turn that runs without one. +"""Generate either a minimal Task Harness or a rich Persona Harness. + +The two modes intentionally have separate prompts, structured inputs, and tool +schemas. Task selects existing agents with only a reusable prompt and suggested +tools; Persona may use the complete enabled Harness surface, including generated +participant functions. Neither generator emits a Workflow. One repair round is +allowed, after which the ordinary unconfigured turn continues. """ from __future__ import annotations import json -from typing import TYPE_CHECKING, Any +import textwrap +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any, Literal from loguru import logger +from raven.agent.harness_capabilities import function_enabled, parameter_enabled from raven.agent.subagent.delegate import DelegateTable, Worker from raven.playbook.agent_spec import AgentPlaybookSpec @@ -40,19 +34,56 @@ the author wrote into one the gate refuses for a reason they cannot see.""" MAX_REPAIR_ROUNDS = 1 -EMIT_TOOL = "emit_worker_table" - -SYSTEM_PROMPT = ( - "You prepare the workers for one task before the agent that will run it starts.\n\n" - "Given the task, decide which of the offered sub-agents it needs and write each one a brief. " - "Two entries may name the same agent with different briefs -- that is how one task gets a " - "worker per subject. Give a worker a distinct `as` label when you do that.\n\n" - "Write a brief only where it changes what the worker would do: what to cover, what to leave " - "alone, where to put its output, what counts as done. Do not restate the agent's own job -- it " - "already knows that. If the task needs no sub-agents, emit an empty list.\n\n" - "Be sparing. Every worker you name is a separate process the agent has to wait for." +TASK_TOOL = "create_task_playbook" +PERSONA_TOOL = "create_persona_playbook" +# Compatibility alias for callers of the original task-only generator. +EMIT_TOOL = TASK_TOOL + +HarnessDisposition = Literal["none", "runtime", "artifact", "runtime_and_artifact"] +GenerationMode = Literal["task", "persona"] + + +@dataclass(frozen=True) +class HarnessResolution: + """A validated generated Harness ready for binding and persistence.""" + + disposition: HarnessDisposition = "none" + spec: AgentPlaybookSpec | None = None + table: DelegateTable | None = None + selected_playbook: str | None = None + artifact_name: str | None = None + capture_workflow: bool = False + active: bool = False + description: str = "" + generation_mode: GenerationMode | None = None + persisted: bool = False + + +TASK_SYSTEM_PROMPT = ( + "Create a reusable task Playbook Harness for the request. Call create_task_playbook and emit no prose.\n\n" + "Treat the user text as one task instance, not a persona specification. Select registered agents by their " + "declared ownership and capabilities; a generic agent must not absorb work explicitly owned by a specialist. " + "Each worker has exactly one reusable prompt plus optional suggested tools. Keep run-specific companies, " + "dates, destinations, and other values out of that durable prompt: the actual task is injected separately " + "when spawn or a DAG node dispatches the worker. Do not invent a DAG here. The main agent plans and executes " + "normally, and the host later captures a complete successful DAG as the Workflow. Use no fields beyond the " + "offered schema." ) +PERSONA_SYSTEM_PROMPT = ( + "Create a durable digital-person Harness. Call create_persona_playbook and emit no prose.\n\n" + "Treat the user text as requirements for the persona's future capabilities, behavior, style, and constraints. " + "Workers are operating components of the persona, not temporary authors asked to design or save it. Write " + "every brief, system prompt, stop condition, check, and function for future runtime behavior; remove creation-" + "time commands. Choose distinct specialist owners for materially different responsibilities, and never let a " + "generic agent absorb work a specialist explicitly owns. Do not design a Workflow or invent a DAG for the act " + "of creating the persona. Generate participant functions only for concrete long-lived runtime rules that " + "ordinary instructions cannot enforce, following the supplied contracts and participantFunctionSyntax exactly. " + "Never use for/while statements, try/except, imports, append, or an unlisted call in generated functions." +) + +# Compatibility for direct imports that historically meant the task generator. +SYSTEM_PROMPT = TASK_SYSTEM_PROMPT _RULE_SCHEMA: dict[str, Any] = { "type": "object", @@ -80,10 +111,212 @@ Spelled out here rather than derived from :class:`CheckRule`: a schema the model reads wants a sentence per field saying when to write it, and a dump of the dataclass would carry the field names without the reason for any of them. -The two are held together by :func:`_spec_from_args`, which validates what +The two are held together by :func:`_persona_spec_from_args`, which validates what comes back against the real model. """ +FUNCTION_REFERENCE_IMPLEMENTATIONS: dict[str, str] = { + "intake": "def intake(text, step):\n return None", + "advise": "def advise(step):\n return None", + "judge": "def judge(name, params, prior):\n return []", + "salvage": "def salvage(step):\n return None", +} +"""The generated, synchronous equivalents of AgentParticipant's default no-ops.""" + + +_STEP_FIELDS = ( + "session_key, iteration, response, transcript, history, turn_base, question, " + "rollbacks, mode, mode_overlay, phase, tools, window, max_iterations, tools_ran" +) + + +_FUNCTION_CONTRACTS: dict[str, dict[str, str]] = { + "intake": { + "when": "Runs once on inbound user text, before any model call.", + "returns": ( + "None for no opinion; otherwise a dict. text replaces the inbound text, reply ends the turn " + "before a model call, and note is diagnostic. Across participants, text changes are threaded " + "in order and the first non-null reply stops intake." + ), + }, + "advise": { + "when": "Runs around model calls with the current read-only step.", + "returns": ( + "None for no opinion or a guidance string. Guidance from all participants is joined in order " + "with a blank line." + ), + }, + "judge": { + "when": "Runs synchronously before each worker tool call.", + "returns": ( + "An empty list to allow the call, or refusal sentences returned to the worker so it can retry. " + "The first participant with a non-empty refusal decides." + ), + }, + "salvage": { + "when": "Runs only when a turn otherwise ended without a final answer.", + "returns": "None for no opinion or a final reply string. The first non-empty string decides.", + }, +} + + +def participant_function_guide() -> dict[str, dict[str, str]]: + """The exact runtime contract and executable no-op each generated hook extends.""" + guide: dict[str, dict[str, str]] = {} + for module, name in ( + ("memory", "intake"), + ("planning", "advise"), + ("action", "judge"), + ("action", "salvage"), + ): + if not function_enabled(module, "participant", name): + continue + guide[name] = { + "signature": FUNCTION_REFERENCE_IMPLEMENTATIONS[name].splitlines()[0], + "defaultImplementation": FUNCTION_REFERENCE_IMPLEMENTATIONS[name], + "when": _FUNCTION_CONTRACTS[name]["when"], + "returnsAndComposition": _FUNCTION_CONTRACTS[name]["returns"], + "availableInputs": ( + "step is a read-only JSON-like dict with " + _STEP_FIELDS + if name != "judge" + else "name is the tool name; params is its argument dict; prior is [(tool_name, params), ...]." + ), + } + return guide + + +def participant_function_syntax_guide() -> dict[str, Any]: + """The executable subset enforced by charter_code, in model-facing terms.""" + return { + "shape": "Exactly one synchronous def with the supplied signature; no statements outside it.", + "allowedStatements": ["assignment", "if", "return"], + "allowedIteration": "Use a list or generator comprehension; for and while statements are forbidden.", + "allowedCalls": [ + "len", + "str", + "int", + "float", + "bool", + "isinstance", + "any", + "all", + "sorted", + "set", + "list", + "dict", + "tuple", + "min", + "max", + "abs", + ], + "allowedMethods": [ + "startswith", + "endswith", + "lower", + "upper", + "strip", + "split", + "get", + "items", + "keys", + "values", + "count", + "join", + ], + "forbidden": [ + "for or while statements", + "try/except", + "imports", + "with", + "raise", + "lambda", + "async/await", + "append or any unlisted call or method", + ], + } + + +def _choice_profile(choice: Any) -> dict[str, str]: + """A mode/model choice without transport-owned or secret configuration.""" + read = choice.get if isinstance(choice, dict) else lambda key, default="": getattr(choice, key, default) + profile = { + "id": str(read("id", "") or read("value", "") or ""), + "name": str(read("name", "") or read("label", "") or ""), + "description": str(read("description", "") or ""), + } + return {key: value for key, value in profile.items() if value} + + +def task_roster_profile(meta: Any) -> dict[str, Any]: + """Only the safe AgentMeta facts the task selector can act on.""" + return { + "description": str(getattr(meta, "description", "") or ""), + "owns": str(getattr(meta, "owns", "") or ""), + "stateful": bool(getattr(meta, "stateful", False)), + "readsLocalFiles": bool(getattr(meta, "reads_local_files", False)), + "liveProgress": bool(getattr(meta, "live_progress", False)), + "ownsWatchedWork": bool(getattr(meta, "owns_watched_work", False)), + } + + +def persona_roster_profile(meta: Any) -> dict[str, Any]: + """Every safe AgentMeta fact that can shape a durable persona.""" + return { + "description": str(getattr(meta, "description", "") or ""), + "owns": str(getattr(meta, "owns", "") or ""), + "stateful": bool(getattr(meta, "stateful", False)), + "readsLocalFiles": bool(getattr(meta, "reads_local_files", False)), + "liveProgress": bool(getattr(meta, "live_progress", False)), + "ownsWatchedWork": bool(getattr(meta, "owns_watched_work", False)), + "modes": [profile for choice in (getattr(meta, "modes", ()) or ()) if (profile := _choice_profile(choice))], + "modelChoices": [ + profile for choice in (getattr(meta, "model_choices", ()) or ()) if (profile := _choice_profile(choice)) + ], + } + + +# Compatibility for callers that asked for the formerly single rich profile. +roster_profile = persona_roster_profile + + +def _profile_summary(value: Any) -> str: + if isinstance(value, str): + return value.strip() + if not isinstance(value, dict): + return "" + prose = str(value.get("owns") or value.get("description") or "").strip().rstrip(".") + nested = value.get("capabilities") if isinstance(value.get("capabilities"), dict) else {} + caps = { + **nested, + **{ + key: value[key] + for key in ("readsLocalFiles", "stateful", "liveProgress", "ownsWatchedWork") + if key in value + }, + } + tags = [ + label + for key, label in ( + ("readsLocalFiles", "reads local files"), + ("stateful", "resumable"), + ("liveProgress", "reports live progress"), + ("ownsWatchedWork", "owns watched work"), + ) + if caps.get(key) + ] + if prose and tags: + return f"{prose} ({'; '.join(tags)})" + return prose or "; ".join(tags) + + +def _profile_payload(value: Any) -> dict[str, Any]: + """Normalize the structured profile while tolerating the old prose form.""" + if isinstance(value, dict): + return dict(value) + if isinstance(value, str) and value.strip(): + return {"description": value.strip()} + return {} + def roster_note(meta: Any) -> str: """One line on what an agent is for, in the terms a choice between two turns on. @@ -112,7 +345,7 @@ def roster_note(meta: Any) -> str: return prose or "; ".join(tags) -def _roster_description(agent_names: list[str], agent_notes: "Mapping[str, str] | None") -> str: +def _roster_description(agent_names: list[str], agent_notes: "Mapping[str, Any] | None") -> str: """What each name on the roster is for, or the bare instruction without them. An enum of names alone is only selectable when the names say what they are. @@ -125,15 +358,84 @@ def _roster_description(agent_names: list[str], agent_notes: "Mapping[str, str] """ head = "Which sub-agent this worker is." lines = [ - f"- {name}: {agent_notes[name].strip()}" for name in agent_names if (agent_notes or {}).get(name, "").strip() + f"- {name}: {summary}" for name in agent_names if (summary := _profile_summary((agent_notes or {}).get(name))) ] return head if not lines else head + " What each one is for:\n" + "\n".join(lines) -def emit_tool( - agent_names: list[str], tool_names: list[str], agent_notes: "Mapping[str, str] | None" = None +def task_tool(agent_names: list[str], tool_names: list[str]) -> list[dict[str, Any]]: + """The deliberately small task selector: agent, reusable prompt, and tools.""" + worker = { + "type": "object", + "properties": { + "as": { + "type": "string", + "minLength": 1, + "description": "Optional local label; omit when the registered agent name is sufficient.", + }, + "agent": { + "type": "string", + "enum": agent_names, + "description": "The registered sub-agent that owns this part of the task.", + }, + "prompt": { + "type": "string", + "minLength": 1, + "description": ( + "Reusable responsibilities and instructions for this worker. Do not include values " + "specific to this run; the dispatch supplies the actual task separately." + ), + }, + "tools": { + "type": "array", + "items": {"type": "string", "enum": tool_names}, + "description": "Optional tools this worker's role calls for; guidance, not permission.", + }, + }, + "required": ["agent", "prompt"], + "additionalProperties": False, + } + if not parameter_enabled("capability", "tools"): + worker["properties"].pop("tools") + return [ + { + "type": "function", + "function": { + "name": TASK_TOOL, + "description": ( + "Create and save the minimal worker selection for a reusable task. Do not emit a DAG " + "or any Persona-only Harness field." + ), + "parameters": { + "type": "object", + "properties": { + "artifactName": { + "type": "string", + "pattern": "^[a-z0-9][a-z0-9-]*$", + "description": "Durable kebab-case Playbook name.", + }, + "description": { + "type": "string", + "minLength": 1, + "maxLength": 200, + "description": "One reusable sentence describing the task Playbook.", + }, + "workers": {"type": "array", "items": worker, "minItems": 1}, + }, + "required": ["artifactName", "description", "workers"], + "additionalProperties": False, + }, + }, + } + ] + + +def persona_tool( + agent_names: list[str], + tool_names: list[str], + agent_notes: "Mapping[str, Any] | None" = None, ) -> list[dict[str, Any]]: - """The one tool the generator may call, with the install's own enums. + """The rich Persona Harness tool, with the install's own enums. The roster is an ``enum`` and the tool list is an ``enum`` for the same reason: a name the host cannot resolve is worth refusing at the boundary @@ -144,23 +446,22 @@ def emit_tool( "properties": { "as": { "type": "string", - "description": ( - "Optional short label for this worker (e.g. 'research-a'). Give one only when " - "the same agent appears more than once; otherwise omit it." - ), + "minLength": 1, + "description": ("Short stable label for this persona component (for example 'research-a')."), }, - "name": { + "agent": { "type": "string", "enum": agent_names, "description": _roster_description(agent_names, agent_notes), }, "brief": { "type": "string", + "minLength": 1, "description": "One line on what this worker is for, read by the agent that dispatches it.", }, "systemPrompt": { "type": "string", - "description": "What this worker should and should not do. Omit when the agent's own job covers it.", + "description": "Durable persona instructions appended to the worker's existing identity. Omit when the brief is enough.", }, "stopWhen": {"type": "string", "description": "Optional: what, once obtained, means this worker is done."}, "tools": { @@ -178,21 +479,9 @@ def emit_tool( "cannot hold -- 'stay under this directory', 'read it before writing it'." ), }, - "code": { - "type": "string", - "description": ( - "Optional: a judgement the rules above cannot state, as Python. Define " - "judge(name, params, prior) returning the refusal sentences for one call " - "(an empty list allows it); prior is [(tool_name, params), ...] of what " - "already ran successfully this turn. An allow-listed subset only: no " - "imports, no loops, no attribute starting with '_', and no calls beyond " - "len/str/int/bool/any/all/sorted/set/list/dict/tuple/min/max/abs and the " - "string and dict methods. Comprehensions are allowed and are how you walk " - "prior. Source outside the subset is dropped, so keep it to one judgement." - ), - }, "timeoutSeconds": { "type": "integer", + "minimum": 1, "description": ( "Optional: a deadline for this worker, in seconds. Tightening only -- it " "cannot lengthen a limit an operator set. Give one only when the job is " @@ -200,29 +489,104 @@ def emit_tool( ), }, }, - "required": ["name"], + "required": ["as", "agent", "brief"], "additionalProperties": False, } - return [ - { + properties = worker["properties"] + for module, name, field in ( + ("memory", "systemPrompt", "systemPrompt"), + ("memory", "stopWhen", "stopWhen"), + ("capability", "tools", "tools"), + ("action", "checks", "checks"), + ): + if not parameter_enabled(module, name): + properties.pop(field, None) + function_properties = { + name: { + "type": "string", + "description": participant_function_guide()[name]["returnsAndComposition"] + + " Use the exact signature in the supplied participantFunctions input and the allow-listed Python subset.", + } + for module, name in ( + ("memory", "intake"), + ("planning", "advise"), + ("action", "judge"), + ("action", "salvage"), + ) + if function_enabled(module, "participant", name) + } + if function_properties: + properties["functions"] = { + "type": "object", + "properties": function_properties, + "additionalProperties": False, + "description": "Optional generated participant functions, keyed by their loop verb.", + } + description = { + "type": "string", + "minLength": 1, + "maxLength": 200, + "description": "One sentence explaining the generated setup.", + } + artifact_name = { + "type": "string", + "pattern": "^[a-z0-9][a-z0-9-]*$", + "description": "Durable kebab-case Playbook name.", + } + + def decision_tool( + name: str, + summary: str, + schema_properties: dict[str, Any], + required: list[str], + ) -> dict[str, Any]: + return { "type": "function", "function": { - "name": EMIT_TOOL, - "description": "Emit the workers this task needs.", + "name": name, + "description": summary, "parameters": { "type": "object", - "properties": { - "description": {"type": "string", "description": "One sentence on the plan (<= 200 chars)."}, - "workers": {"type": "array", "items": worker}, - }, - "required": ["workers"], + "properties": schema_properties, + "required": required, "additionalProperties": False, }, }, } + + return [ + decision_tool( + PERSONA_TOOL, + ( + "Create a durable digital person, assistant, or agent Harness. Workers are the persona's " + "future operating parts, not authors asked to create it. Write future-facing briefs, system " + "prompts, tools, functions, checks, and stop conditions; remove creation-time commands such " + "as save it, do not run now, or only plan later. A single named persona may still use several " + "workers when different offered specialists own materially different behavior. Use intake " + "for an enforceable missing-input gate, not merely to repeat requirements. This branch never " + "captures a Workflow." + ), + { + "description": description, + "artifactName": artifact_name, + "workers": {"type": "array", "items": worker, "minItems": 1}, + }, + ["artifactName", "description", "workers"], + ), ] +def emit_tool( + agent_names: list[str], + tool_names: list[str], + agent_notes: "Mapping[str, Any] | None" = None, + playbook_candidates: "Mapping[str, str] | None" = None, +) -> list[dict[str, Any]]: + """Backward-compatible name for the task-only tool schema.""" + del agent_notes, playbook_candidates + return task_tool(agent_names, tool_names) + + def render_charter(brief: str, system_prompt: str, stop_when: str, tools: list[str] | None) -> str: """The preamble one worker's task carries. @@ -231,13 +595,13 @@ def render_charter(brief: str, system_prompt: str, stop_when: str, tools: list[s drift. """ lines: list[str] = [] - if system_prompt.strip(): - lines.append(system_prompt.strip()) - elif brief.strip(): + if brief.strip(): lines.append(brief.strip()) - if tools: + if parameter_enabled("memory", "systemPrompt") and system_prompt.strip(): + lines.append(system_prompt.strip()) + if parameter_enabled("capability", "tools") and tools: lines.append(f"Tools this job calls for: {', '.join(tools)}.") - if stop_when.strip(): + if parameter_enabled("memory", "stopWhen") and stop_when.strip(): lines.append(f"Done when: {stop_when.strip()}") if not lines: return "" @@ -253,24 +617,43 @@ def build_payload(brief: str, spec: "SubPlaybook | None") -> dict[str, Any] | No preamble. Both are built from one source so the two can never say different things. """ - prompt = (spec.memory.system_prompt if spec else "") or brief - tools = spec.capability.tools if spec else None + instruction_addendum = ( + (spec.memory.system_prompt if spec else "") if parameter_enabled("memory", "systemPrompt") else "" + ) + tools = spec.capability.tools if spec and parameter_enabled("capability", "tools") else None checks = spec.action.checks if spec and spec.action.checks else None - stop_when = spec.stop_when if spec else "" + stop_when = spec.stop_when if spec and parameter_enabled("memory", "stopWhen") else "" payload: dict[str, Any] = {} - if prompt.strip(): - payload["prompt"] = prompt.strip() + if brief.strip(): + payload["brief"] = brief.strip() + if instruction_addendum.strip(): + payload["instructionAddendum"] = instruction_addendum.strip() + legacy_prompt = "\n\n".join(part for part in (brief.strip(), instruction_addendum.strip()) if part) + if legacy_prompt: + payload["prompt"] = legacy_prompt if tools is not None: payload["tools"] = list(tools) if stop_when.strip(): payload["stopWhen"] = stop_when.strip() - if checks and checks.rules: + if parameter_enabled("action", "checks") and checks and checks.rules: payload["checks"] = [rule.model_dump(by_alias=True, exclude_defaults=True) for rule in checks.rules] # Carried whether or not there are rules beside it: a judgement some jobs # can only state as code is the reason the field exists, and one written # without any declarative rule would otherwise be dropped on the way out. - if checks and checks.code.strip(): + if function_enabled("action", "participant", "judge") and checks and checks.code.strip(): payload["code"] = checks.code + generated_functions: dict[str, str] = {} + if spec: + for module, functions in ( + ("memory", spec.memory.functions), + ("planning", spec.planning.functions), + ("action", spec.action.functions), + ): + for name, source in functions.items(): + if function_enabled(module, "participant", name) and source.strip(): + generated_functions[name] = source + if generated_functions: + payload["functions"] = generated_functions if spec and spec.timeout_seconds: payload["timeoutSeconds"] = spec.timeout_seconds return payload or None @@ -281,7 +664,7 @@ def build_table(spec: AgentPlaybookSpec, briefs: dict[str, str]) -> DelegateTabl workers: dict[str, Worker] = {} for entry in spec.delegate: pb = entry.playbook - brief = briefs.get(entry.label, "") + brief = briefs.get(entry.label, entry.brief) charter = render_charter( brief, pb.memory.system_prompt if pb else "", @@ -298,11 +681,74 @@ def build_table(spec: AgentPlaybookSpec, briefs: dict[str, str]) -> DelegateTabl return DelegateTable(workers=workers) -def _spec_from_args(args: dict[str, Any], roster: set[str]) -> tuple[AgentPlaybookSpec, dict[str, str]]: - """Build a spec from the emitted arguments, or raise with what to repair.""" +def _short_description(value: Any) -> str: + """Keep generated index prose readable when a provider misses maxLength.""" + text = " ".join(str(value or "").split()) + return text if len(text) <= 200 else textwrap.shorten(text, width=200, placeholder="…") + + +def _task_spec_from_args( + args: dict[str, Any], + roster: set[str], + available_tools: set[str] | None = None, +) -> tuple[AgentPlaybookSpec, dict[str, str]]: + """Build the minimal Task Harness and reject every Persona-only field.""" + rows = args.get("workers") + if not isinstance(rows, list) or not rows: + raise ValueError("workers must be a non-empty list") + delegate: list[dict[str, Any]] = [] + briefs: dict[str, str] = {} + errors: list[str] = [] + allowed = {"as", "agent", "prompt", "tools"} + for index, row in enumerate(rows, 1): + if not isinstance(row, dict): + errors.append(f"{index}. a worker must be an object") + continue + unexpected = sorted(set(row) - allowed) + if unexpected: + errors.append(f"{index}. Task worker field(s) not allowed: {', '.join(unexpected)}") + continue + name = str(row.get("agent") or "") + prompt = str(row.get("prompt") or "").strip() + if name not in roster: + errors.append(f"{index}. unknown agent: {name or ''}") + continue + if not prompt: + errors.append(f"{index}. prompt must be non-empty") + continue + tools = row.get("tools") + if tools is not None and not isinstance(tools, list): + errors.append(f"{index}. tools must be an array") + continue + if tools is not None and not parameter_enabled("capability", "tools"): + errors.append(f"{index}. disabled harness field(s): tools") + continue + unknown_tools = sorted({str(tool) for tool in tools or []} - (available_tools or set())) + if available_tools is not None and unknown_tools: + errors.append(f"{index}. unknown tool(s): {', '.join(unknown_tools)}") + continue + label = str(row.get("as") or "") or name + sub = {"capability": {"tools": [str(tool) for tool in tools]}} if tools is not None else None + delegate.append({"as": label, "name": name, "brief": prompt, "playbook": sub}) + briefs[label] = prompt + if errors: + raise ValueError("; ".join(errors)) + payload = { + "description": _short_description(args.get("description")), + "delegate": delegate, + } + return AgentPlaybookSpec.model_validate(payload), briefs + + +def _persona_spec_from_args( + args: dict[str, Any], + roster: set[str], + available_tools: set[str] | None = None, +) -> tuple[AgentPlaybookSpec, dict[str, str]]: + """Build a rich Persona Harness, or raise with what to repair.""" rows = args.get("workers") - if not isinstance(rows, list): - raise ValueError("workers must be a list") + if not isinstance(rows, list) or not rows: + raise ValueError("workers must be a non-empty list") delegate: list[dict[str, Any]] = [] briefs: dict[str, str] = {} errors: list[str] = [] @@ -310,66 +756,189 @@ def _spec_from_args(args: dict[str, Any], roster: set[str]) -> tuple[AgentPlaybo if not isinstance(row, dict): errors.append(f"{index}. a worker must be an object") continue - name = str(row.get("name") or "") + allowed = { + "as", + "agent", + "brief", + "systemPrompt", + "stopWhen", + "tools", + "checks", + "functions", + "timeoutSeconds", + } + unexpected = sorted(set(row) - allowed) + if unexpected: + errors.append(f"{index}. Persona worker field(s) not allowed: {', '.join(unexpected)}") + continue + name = str(row.get("agent") or "") if name not in roster: # Dropped rather than repaired: the roster was an enum, so a name # outside it is a shape the request could not express, and spending # a round on it teaches the model nothing it was not already told. logger.info("agent playbook: dropping worker {!r} -- not on the roster", name) continue - label = str(row.get("as") or "") or name + label = str(row.get("as") or "").strip() + brief = str(row.get("brief") or "").strip() + if not label: + errors.append(f"{index}. as must be non-empty") + if not brief: + errors.append(f"{index}. brief must be non-empty") + switches = ( + ("systemPrompt", parameter_enabled("memory", "systemPrompt")), + ("stopWhen", parameter_enabled("memory", "stopWhen")), + ("tools", parameter_enabled("capability", "tools")), + ("checks", parameter_enabled("action", "checks")), + ) + disabled = [field for field, enabled in switches if field in row and not enabled] + if disabled: + errors.append(f"{index}. disabled harness field(s): {', '.join(disabled)}") + continue + raw_functions = row.get("functions") + if "functions" in row and not isinstance(raw_functions, dict): + errors.append(f"{index}. functions must be an object") + continue + function_modules = { + "intake": "memory", + "advise": "planning", + "judge": "action", + "salvage": "action", + } + generated_functions: dict[str, tuple[str, str]] = {} + for function_name, source in (raw_functions or {}).items(): + module = function_modules.get(function_name) + if module is None: + errors.append(f"{index}. unknown generated participant function: {function_name}") + continue + if not function_enabled(module, "participant", function_name): + errors.append(f"{index}. disabled harness field(s): functions.{function_name}") + continue + if not isinstance(source, str) or not source.strip(): + errors.append(f"{index}. functions.{function_name} must be non-empty Python source") + continue + if len(source) > MAX_CODE_CHARS: + errors.append(f"{index}. functions.{function_name} exceeds {MAX_CODE_CHARS} characters") + continue + try: + from raven.agent.subagent.charter_code import compile_function + + compile_function(source, function_name) + except Exception as exc: + errors.append(f"{index}. functions.{function_name} was refused: {exc}") + continue + generated_functions[function_name] = (module, source) + if any(message.startswith(f"{index}.") for message in errors): + continue sub: dict[str, Any] = {} if row.get("systemPrompt"): sub.setdefault("memory", {})["systemPrompt"] = str(row["systemPrompt"]) if row.get("stopWhen"): sub["stopWhen"] = str(row["stopWhen"]) - if isinstance(row.get("tools"), list): - sub.setdefault("capability", {})["tools"] = [str(x) for x in row["tools"]] + for function_name, (module, source) in generated_functions.items(): + if function_name == "judge": + sub.setdefault("action", {}).setdefault("checks", {})["code"] = source + else: + sub.setdefault(module, {}).setdefault("functions", {})[function_name] = source + raw_tools = row.get("tools") + if "tools" in row and not isinstance(raw_tools, list): + errors.append(f"{index}. tools must be an array") + continue + if isinstance(raw_tools, list): + requested_tools = [str(x) for x in raw_tools] + unknown_tools = sorted(set(requested_tools) - (available_tools or set())) + if available_tools is not None and unknown_tools: + errors.append(f"{index}. unknown tool(s): {', '.join(unknown_tools)}") + continue + sub.setdefault("capability", {})["tools"] = requested_tools # Both halves of the judgement seat land under one key, because a # worker with code and no rules is as ordinary as one with rules and no # code -- the two are alternatives, not a base and an extension. - checks: dict[str, Any] = {} - if isinstance(row.get("checks"), list) and row["checks"]: - checks["rules"] = row["checks"] - if row.get("code"): - checks["code"] = str(row["code"])[:MAX_CODE_CHARS] + checks: dict[str, Any] = dict(sub.get("action", {}).get("checks", {})) + raw_checks = row.get("checks") + if "checks" in row and not isinstance(raw_checks, list): + errors.append(f"{index}. checks must be an array") + continue + if isinstance(raw_checks, list) and raw_checks: + checks["rules"] = raw_checks if checks: sub.setdefault("action", {})["checks"] = checks timeout = row.get("timeoutSeconds") - if isinstance(timeout, int) and not isinstance(timeout, bool) and timeout > 0: + if "timeoutSeconds" in row: + if not isinstance(timeout, int) or isinstance(timeout, bool) or timeout <= 0: + errors.append(f"{index}. timeoutSeconds must be a positive integer") + continue sub["timeoutSeconds"] = timeout - delegate.append({"as": label, "name": name, "playbook": sub or None}) - briefs[label] = str(row.get("brief") or "") + delegate.append({"as": label, "name": name, "brief": brief, "playbook": sub or None}) + briefs[label] = brief if errors: raise ValueError("; ".join(errors)) payload = {"delegate": delegate} if args.get("description"): - payload["description"] = str(args["description"])[:200] + payload["description"] = _short_description(args["description"]) return AgentPlaybookSpec.model_validate(payload), briefs -class WorkerTableGenerator: - """One model call that writes this turn's worker table.""" +def _tool_inventory(tool_catalog: list[Any]) -> list[dict[str, str]]: + inventory: list[dict[str, str]] = [] + for item in tool_catalog: + if isinstance(item, str): + name, description = item, "" + elif isinstance(item, dict): + function = item.get("function", item) + if not isinstance(function, dict): + continue + name = str(function.get("name") or "") + description = str(function.get("description") or "") + else: + continue + if name: + inventory.append({"name": name, "description": description}) + return sorted(inventory, key=lambda item: item["name"]) + + +def persona_parameter_guide() -> dict[str, str]: + """The Persona-only Harness parameters and their runtime meaning.""" + guide = { + "brief": "The worker's durable responsibility.", + "systemPrompt": "Long-lived behavior and style appended to the worker identity.", + "stopWhen": "What means this worker is finished.", + "tools": "Suggested tools for this role; guidance, not permission.", + "checks": "Declarative rules evaluated before worker tool calls.", + "functions": "Generated Participant runtime functions.", + "timeoutSeconds": "A tightening-only dispatch deadline.", + } + switches = { + "systemPrompt": parameter_enabled("memory", "systemPrompt"), + "stopWhen": parameter_enabled("memory", "stopWhen"), + "tools": parameter_enabled("capability", "tools"), + "checks": parameter_enabled("action", "checks"), + } + return {name: detail for name, detail in guide.items() if switches.get(name, True)} + + +class _PlaybookGenerator: + mode: GenerationMode + prompt: str + tool_name: str def __init__(self, provider: "LLMProvider", model: str | None = None) -> None: self._provider = provider self._model = model - async def generate( + async def _resolve( self, + *, query: str, agent_names: list[str], - tool_names: list[str], - agent_notes: "Mapping[str, str] | None" = None, - ) -> DelegateTable | None: - """The table for ``query``, or ``None`` to run this turn unconfigured.""" + user_payload: dict[str, Any], + tools: list[dict[str, Any]], + ) -> HarnessResolution: if not query.strip() or not agent_names: - return None + return HarnessResolution() messages: list[dict[str, Any]] = [ - {"role": "system", "content": SYSTEM_PROMPT}, - {"role": "user", "content": f"Task:\n{query}"}, + {"role": "system", "content": self.prompt}, + {"role": "user", "content": json.dumps(user_payload, ensure_ascii=False, indent=2)}, ] - tools = emit_tool(sorted(agent_names), sorted(tool_names), agent_notes) roster = set(agent_names) for _ in range(1 + MAX_REPAIR_ROUNDS): try: @@ -377,27 +946,136 @@ async def generate( messages=messages, tools=tools, model=self._model or None, - tool_choice={"type": "function", "function": {"name": EMIT_TOOL}}, + tool_choice={"type": "function", "function": {"name": self.tool_name}}, ) - except Exception as exc: # noqa: BLE001 - a failed setup step must not cost the turn - logger.warning("agent playbook: generation call failed ({}); running unconfigured", exc) - return None - args = _emitted_args(response) + except Exception as exc: # noqa: BLE001 - setup failure must not cost the turn + logger.warning("agent playbook: {} generation failed ({}); running unconfigured", self.mode, exc) + return HarnessResolution() + args = _emitted_args(response, self.tool_name) if args is None: - messages.append({"role": "user", "content": f"You emitted no table. Call {EMIT_TOOL}."}) + messages.append({"role": "user", "content": f"Call {self.tool_name} with a valid payload."}) continue try: - spec, briefs = _spec_from_args(args, roster) + if self.mode == "task": + spec, briefs = _task_spec_from_args( + args, roster, {item["name"] for item in user_payload["availableTools"]} + ) + else: + spec, briefs = _persona_spec_from_args( + args, + roster, + {item["name"] for item in user_payload["availableTools"]}, + ) + from raven.playbook.types import slugify + + description = _short_description(args.get("description") or spec.description or query.splitlines()[0]) + if not description: + raise ValueError("description must be non-empty") + artifact_name = slugify(str(args.get("artifactName") or description)) + spec = spec.model_copy(update={"name": artifact_name, "description": description}) + table = build_table(spec, briefs) except Exception as exc: # noqa: BLE001 - the message is the repair prompt - logger.info("agent playbook: rejected, repairing once ({})", exc) - messages.append({"role": "user", "content": f"That table was rejected: {exc}\nEmit a corrected one."}) + logger.info("agent playbook: rejected {} Harness, repairing once ({})", self.mode, exc) + messages.append({"role": "user", "content": f"That Harness was rejected: {exc}\nEmit a corrected one."}) continue - return build_table(spec, briefs) or None - logger.warning("agent playbook: still invalid after one repair; running unconfigured") - return None + return HarnessResolution( + disposition="runtime_and_artifact" if self.mode == "task" else "artifact", + active=True, + spec=spec, + table=table, + artifact_name=artifact_name, + capture_workflow=self.mode == "task", + description=description, + generation_mode=self.mode, + ) + logger.warning("agent playbook: {} Harness still invalid after one repair; running unconfigured", self.mode) + return HarnessResolution() + + +class TaskPlaybookGenerator(_PlaybookGenerator): + """Select existing agents with only a reusable prompt and suggested tools.""" + + mode: GenerationMode = "task" + prompt = TASK_SYSTEM_PROMPT + tool_name = TASK_TOOL + + async def resolve( + self, + query: str, + agent_names: list[str], + tool_catalog: list[Any], + agent_profiles: "Mapping[str, Any] | None" = None, + playbook_candidates: "Mapping[str, str] | None" = None, + ) -> HarnessResolution: + del playbook_candidates # Retrieval is explicitly outside this version. + inventory = _tool_inventory(tool_catalog) + payload = { + "task": query, + "agents": [ + {"name": name, **_profile_payload((agent_profiles or {}).get(name))} for name in sorted(agent_names) + ], + "availableTools": inventory, + } + return await self._resolve( + query=query, + agent_names=agent_names, + user_payload=payload, + tools=task_tool(sorted(agent_names), [item["name"] for item in inventory]), + ) + + async def generate( + self, + query: str, + agent_names: list[str], + tool_catalog: list[Any], + agent_profiles: "Mapping[str, Any] | None" = None, + ) -> DelegateTable | None: + return (await self.resolve(query, agent_names, tool_catalog, agent_profiles)).table + + +class PersonaPlaybookGenerator(_PlaybookGenerator): + """Generate the full durable Harness surface for a digital person.""" + + mode: GenerationMode = "persona" + prompt = PERSONA_SYSTEM_PROMPT + tool_name = PERSONA_TOOL + + async def resolve( + self, + query: str, + agent_names: list[str], + tool_catalog: list[Any], + agent_profiles: "Mapping[str, Any] | None" = None, + existing_artifact_names: list[str] | None = None, + ) -> HarnessResolution: + inventory = _tool_inventory(tool_catalog) + profiles = {name: _profile_payload((agent_profiles or {}).get(name)) for name in sorted(agent_names)} + payload = { + "personaRequirements": query, + "agentProfiles": profiles, + "availableTools": inventory, + "harnessParameters": persona_parameter_guide(), + "participantFunctions": participant_function_guide(), + "participantFunctionSyntax": participant_function_syntax_guide(), + "existingArtifactNames": sorted(existing_artifact_names or []), + } + return await self._resolve( + query=query, + agent_names=agent_names, + user_payload=payload, + tools=persona_tool( + sorted(agent_names), + [item["name"] for item in inventory], + profiles, + ), + ) + + +class WorkerTableGenerator(TaskPlaybookGenerator): + """Backward-compatible task-only generator name.""" -def _emitted_args(response: Any) -> dict[str, Any] | None: +def _emitted_args(response: Any, expected_tool: str = EMIT_TOOL) -> dict[str, Any] | None: """The emitted arguments, from either shape a provider hands back. ``LLMResponse.tool_calls`` carries :class:`ToolCallRequest` objects on the @@ -414,7 +1092,7 @@ def _emitted_args(response: Any) -> dict[str, Any] | None: else: name = getattr(call, "name", None) raw = getattr(call, "arguments", None) - if name != EMIT_TOOL: + if name != expected_tool: continue if isinstance(raw, dict): return raw @@ -428,11 +1106,26 @@ def _emitted_args(response: Any) -> dict[str, Any] | None: __all__ = [ "EMIT_TOOL", + "FUNCTION_REFERENCE_IMPLEMENTATIONS", "MAX_REPAIR_ROUNDS", + "PERSONA_SYSTEM_PROMPT", + "PERSONA_TOOL", + "PersonaPlaybookGenerator", + "TASK_SYSTEM_PROMPT", + "TASK_TOOL", + "TaskPlaybookGenerator", "WorkerTableGenerator", "build_payload", "build_table", "emit_tool", + "participant_function_guide", + "participant_function_syntax_guide", + "persona_parameter_guide", + "persona_roster_profile", + "persona_tool", "roster_note", + "roster_profile", "render_charter", + "task_roster_profile", + "task_tool", ] diff --git a/raven/playbook/agent_spec.py b/raven/playbook/agent_spec.py index 44b57b5c1..9904fc9af 100644 --- a/raven/playbook/agent_spec.py +++ b/raven/playbook/agent_spec.py @@ -33,6 +33,7 @@ from pydantic import Field, model_validator +from raven.agent.harness_capabilities import function_enabled, parameter_enabled from raven.playbook.types import CamelBase AGENT_SPEC_VERSION = 1 @@ -112,6 +113,11 @@ class Checks(CamelBase): class SubMemory(CamelBase): system_prompt: str = "" + functions: dict[str, str] = Field(default_factory=dict) + + +class SubPlanning(CamelBase): + functions: dict[str, str] = Field(default_factory=dict) class SubCapability(CamelBase): @@ -123,6 +129,7 @@ class SubCapability(CamelBase): class SubAction(CamelBase): checks: Checks | None = None + functions: dict[str, str] = Field(default_factory=dict) class SubPlaybook(CamelBase): @@ -130,6 +137,7 @@ class SubPlaybook(CamelBase): role: Literal["subagent"] = "subagent" memory: SubMemory = Field(default_factory=SubMemory) + planning: SubPlanning = Field(default_factory=SubPlanning) capability: SubCapability = Field(default_factory=SubCapability) action: SubAction = Field(default_factory=SubAction) stop_when: str = "" @@ -144,6 +152,39 @@ class SubPlaybook(CamelBase): set; it may not lengthen one an operator set.""" """One line naming what, once obtained, means this worker is done.""" + @model_validator(mode="after") + def _enabled_fields_only(self) -> "SubPlaybook": + disabled: list[str] = [] + if "system_prompt" in self.memory.model_fields_set and not parameter_enabled("memory", "systemPrompt"): + disabled.append("memory.systemPrompt") + if "stop_when" in self.model_fields_set and not parameter_enabled("memory", "stopWhen"): + disabled.append("memory.stopWhen") + if "tools" in self.capability.model_fields_set and not parameter_enabled("capability", "tools"): + disabled.append("capability.tools") + if self.action.checks: + if "rules" in self.action.checks.model_fields_set and not parameter_enabled("action", "checks"): + disabled.append("action.checks") + if "impl" in self.action.checks.model_fields_set and not parameter_enabled("action", "checksImpl"): + disabled.append("action.checksImpl") + if "code" in self.action.checks.model_fields_set and not function_enabled("action", "participant", "judge"): + disabled.append("action.functions.judge") + generated = ( + ("memory", self.memory.functions, {"intake"}), + ("planning", self.planning.functions, {"advise"}), + ("action", self.action.functions, {"salvage"}), + ) + for module, functions, known in generated: + for name, source in functions.items(): + if name not in known: + raise ValueError(f"unknown generated participant function: {module}.{name}") + if not isinstance(source, str) or not source.strip(): + raise ValueError(f"generated participant function is empty: {module}.{name}") + if not function_enabled(module, "participant", name): + disabled.append(f"{module}.functions.{name}") + if disabled: + raise ValueError(f"disabled harness field(s): {', '.join(disabled)}") + return self + @model_validator(mode="after") def _no_third_level(self) -> "SubPlaybook": """A worker may not bring workers of its own. @@ -169,6 +210,9 @@ class DelegateEntry(CamelBase): """The roster agent behind the label. Validated against the live roster by the generator's caller, not here: the schema cannot know the install.""" + brief: str = "" + """The reusable dispatch brief (v1 kept this only in a transient sidecar).""" + playbook: SubPlaybook | None = None @property @@ -204,5 +248,6 @@ def _labels_are_unique(self) -> "AgentPlaybookSpec": "SubAction", "SubCapability", "SubMemory", + "SubPlanning", "SubPlaybook", ] diff --git a/raven/playbook/generator.py b/raven/playbook/generator.py index 474eb92c7..439bb021d 100644 --- a/raven/playbook/generator.py +++ b/raven/playbook/generator.py @@ -180,7 +180,13 @@ def __init__( self._inventory = inventory self._model = model - async def generate(self, user_input: str, skills: list[str] | None = None) -> GeneratedPlaybook: + async def generate( + self, + user_input: str, + skills: list[str] | None = None, + *, + dag_only: bool = False, + ) -> GeneratedPlaybook: """One draft from user input plus optional pinned skills. ``skills`` entries are names, or paths to a skill file whose content @@ -198,14 +204,22 @@ async def generate(self, user_input: str, skills: list[str] | None = None) -> Ge inline_skill_docs=inline_docs, ) known_skills = [name for name, _ in candidates] + pinned + [name for name, _ in inline_docs] + system = SYSTEM_PROMPT + if dag_only: + system += ( + "\n\n# This creation target is Playbook v2\n" + "Emit mode dag and concrete nodes only. Prompt mode is legacy and may be read but is not created. " + "When the user gave no steps, plan the smallest faithful DAG yourself; do not add filler steps." + ) return await self._loop( messages=[ - {"role": "system", "content": SYSTEM_PROMPT}, + {"role": "system", "content": system}, {"role": "user", "content": user_msg}, ], known_skills=known_skills, fixed_name=None, agent_profiles=profiles, + require_dag=dag_only, ) async def revise(self, spec: PlaybookSpec, user_feedback: str) -> GeneratedPlaybook: @@ -237,6 +251,7 @@ async def _loop( known_skills: list[str], fixed_name: str | None, agent_profiles: dict[str, "PlaybookAgentProfile"], + require_dag: bool = False, ) -> GeneratedPlaybook: errors: list[str] = [] for round_no in range(1 + _MAX_REPAIR_ROUNDS): @@ -249,6 +264,8 @@ async def _loop( args = _required_tool_args(response) spec, errors, missing, reported = self._check(args, known_skills, fixed_name, agent_profiles) + if spec is not None and require_dag and spec.mode != "dag": + errors.append("mode: new Playbook v2 artifacts require a concrete dag; prompt mode is legacy-only") if spec is not None and not errors: # The model proposes the L1 vocabulary; the guards decide what # is indexable. Without this the schema hands the model a diff --git a/raven/playbook/harness_generation.json b/raven/playbook/harness_generation.json new file mode 100644 index 000000000..8c3dcffe4 --- /dev/null +++ b/raven/playbook/harness_generation.json @@ -0,0 +1,75 @@ +{ + "harnessGeneration": { + "note": "Controls the fields and functions emitted for a generated worker harness.", + "memory": { + "note": "Controls worker input, context additions, completion, and turn records.", + "parameters": {"note": "Data-only Memory settings.", "items": { + "systemPrompt": {"enabled": true, "note": "Adds task-specific worker instructions."}, + "stopWhen": {"enabled": true, "note": "Names the worker completion condition."}, + "dropSegments": {"enabled": false, "note": "Would omit selected context segments."}, + "memoryTopK": {"enabled": false, "note": "Would limit recalled memory items."} + }}, + "functions": {"note": "Memory functions grouped by execution seat.", + "self": {"note": "Host-owned functions returning or mutating runtime objects.", "items": { + "owns_compaction": {"enabled": false, "note": "Answers whether Memory owns compaction."}, + "candidate_messages": {"enabled": false, "note": "Chooses candidate history messages."}, + "token_budget": {"enabled": false, "note": "Computes the context token budget."}, + "assemble": {"enabled": false, "note": "Builds the host AssembledContext."}, + "shrink": {"enabled": false, "note": "Builds the host ShrinkResult."}, + "after_turn": {"enabled": false, "note": "Updates host memory after a turn."} + }}, + "participant": {"note": "Pure-data functions called through Memory participant seats.", "items": { + "intake": {"enabled": true, "note": "May reshape inbound text or end before a model call."}, + "system_addendum": {"enabled": false, "note": "May add text after the system prefix."}, + "archive": {"enabled": false, "note": "May stamp data on the completed turn record."} + }} + } + }, + "planning": { + "note": "Controls guidance and planning around model calls.", + "parameters": {"note": "Data-only Planning settings.", "items": { + "reasoningEffort": {"enabled": false, "note": "Would override reasoning effort for this dispatch."}, + "maxToolIterations": {"enabled": false, "note": "Would cap tool iterations for this dispatch."} + }}, + "functions": {"note": "Planning functions grouped by execution seat.", + "self": {"note": "Host-owned functions returning runtime planning objects.", "items": { + "prepare": {"enabled": false, "note": "Builds the host PlanningResult."} + }}, + "participant": {"note": "Pure-data functions called through Planning participant seats.", "items": { + "advise": {"enabled": true, "note": "May add guidance around a model call."} + }} + } + }, + "capability": { + "note": "Controls tool definitions offered to the worker.", + "parameters": {"note": "Data-only Capability settings.", "items": { + "tools": {"enabled": true, "note": "Narrows tool names for this dispatch."} + }}, + "functions": {"note": "Capability functions grouped by execution seat.", + "self": {"note": "Host-owned functions returning runtime capability objects.", "items": { + "select": {"enabled": false, "note": "Builds the host CapabilitySelection."} + }}, + "participant": {"note": "Pure-data functions called through Capability participant seats.", "items": { + "select_tools": {"enabled": false, "note": "May reshape the full tool-schema array."} + }} + } + }, + "action": { + "note": "Controls model-call checks and generated judgements.", + "parameters": {"note": "Data-only Action settings.", "items": { + "checks": {"enabled": true, "note": "Adds declarative checks before worker tool calls."}, + "checksImpl": {"enabled": false, "note": "Would select an installed check implementation."} + }}, + "functions": {"note": "Action functions grouped by execution seat.", + "self": {"note": "Host-owned functions performing the model decision.", "items": { + "decide": {"enabled": false, "note": "Builds the host LLMResponse."} + }}, + "participant": {"note": "Pure-data functions called through Action participant seats.", "items": { + "judge": {"enabled": true, "note": "Returns refusal sentences before one tool call."}, + "review": {"enabled": false, "note": "May accept, resample, or end a model step."}, + "salvage": {"enabled": true, "note": "May supply a missing final reply."} + }} + } + } + } +} diff --git a/raven/playbook/run_record.py b/raven/playbook/run_record.py new file mode 100644 index 000000000..68fab4bd4 --- /dev/null +++ b/raven/playbook/run_record.py @@ -0,0 +1,136 @@ +"""Per-turn evidence used to compile and audit durable Playbooks.""" + +from __future__ import annotations + +import hashlib +import json +import os +import tempfile +from contextlib import contextmanager +from contextvars import ContextVar +from dataclasses import dataclass, field +from datetime import UTC, datetime +from pathlib import Path +from typing import Any, Iterator +from uuid import uuid4 + +from raven.agent.subagent.dag_graph import SubAgentDagSpec + + +def _now() -> str: + return datetime.now(UTC).isoformat() + + +@dataclass +class DagEvidence: + run_id: str + spec: SubAgentDagSpec + + +@dataclass +class PlaybookRunCapture: + """The minimum replay evidence from one enabled Playbook turn.""" + + query: str + disposition: str = "none" + selected_playbook: str | None = None + artifact_name: str | None = None + capture_workflow: bool = False + run_id: str = field( + default_factory=lambda: f"pb-{datetime.now(UTC).strftime('%Y%m%dT%H%M%S%fZ')}-{uuid4().hex[:8]}" + ) + started_at: str = field(default_factory=_now) + completed_at: str | None = None + status: str = "running" + dags: list[DagEvidence] = field(default_factory=list) + saved_playbook: str | None = None + error: str | None = None + + def finish(self, *, status: str = "completed", error: str | None = None) -> None: + self.completed_at = _now() + self.status = status + self.error = error + + def as_dict(self) -> dict[str, Any]: + return { + "runId": self.run_id, + "queryDigest": hashlib.sha256(self.query.encode("utf-8")).hexdigest(), + "disposition": self.disposition, + "selectedPlaybook": self.selected_playbook, + "artifactName": self.artifact_name, + "captureWorkflow": self.capture_workflow, + "startedAt": self.started_at, + "completedAt": self.completed_at, + "status": self.status, + "dags": [ + {"runId": dag.run_id, "spec": dag.spec.model_dump(by_alias=True, exclude_none=True)} + for dag in self.dags + ], + "savedPlaybook": self.saved_playbook, + "error": self.error, + } + + +_CAPTURE: ContextVar[PlaybookRunCapture | None] = ContextVar("playbook_run_capture", default=None) + + +@contextmanager +def capture_scope(capture: PlaybookRunCapture | None) -> Iterator[None]: + token = _CAPTURE.set(capture) + try: + yield + finally: + _CAPTURE.reset(token) + + +def current_capture() -> PlaybookRunCapture | None: + """Return the capture object inherited by the current turn task.""" + return _CAPTURE.get() + + +def workflow_capture_requested() -> bool: + """Whether this turn must wait for a successful DAG before promotion.""" + capture = _CAPTURE.get() + return bool(capture is not None and capture.capture_workflow) + + +def record_completed_dag(spec: SubAgentDagSpec, run_id: str, capture: PlaybookRunCapture | None = None) -> None: + """Record a graph only after every node completed successfully.""" + capture = capture or _CAPTURE.get() + if capture is not None: + capture.dags.append(DagEvidence(run_id=run_id, spec=spec)) + + +class RunRecordStore: + """Atomic local run records, deliberately outside the scanned library.""" + + def __init__(self, playbook_root: Path) -> None: + self._root = playbook_root / ".runs" + + def save(self, capture: PlaybookRunCapture) -> Path: + self._root.mkdir(parents=True, exist_ok=True) + path = self._root / f"{capture.run_id}.json" + fd, raw_tmp = tempfile.mkstemp(prefix=f".{path.name}.", dir=self._root) + tmp = Path(raw_tmp) + try: + with os.fdopen(fd, "w", encoding="utf-8") as handle: + json.dump(capture.as_dict(), handle, ensure_ascii=False, indent=2) + handle.write("\n") + handle.flush() + os.fsync(handle.fileno()) + os.chmod(tmp, 0o600) + os.replace(tmp, path) + finally: + tmp.unlink(missing_ok=True) + return path + + +__all__ = [ + "DagEvidence", + "PlaybookRunCapture", + "RunRecordStore", + "capture_scope", + "current_capture", + "record_completed_dag", + "workflow_capture_requested", +] diff --git a/raven/playbook/runtime.py b/raven/playbook/runtime.py index 4ea66887a..4aa87d0d1 100644 --- a/raven/playbook/runtime.py +++ b/raven/playbook/runtime.py @@ -24,6 +24,7 @@ from raven.playbook.store import PlaybookStore from raven.playbook.triggers import find_collisions from raven.playbook.types import PlaybookSpec +from raven.playbook.unified import StoredPlaybook, UnifiedPlaybookSpec from raven.playbook.validate import validate_structure #: How many times one conversation may be told "still missing X" for the same @@ -67,7 +68,7 @@ def __init__( #: the bundled ``create_playbook`` tool binds one object. ``None`` on a #: host that assembles a runtime without creation (the CLI's run path). self._generator = generator - self._specs: dict[str, PlaybookSpec] = {} + self._specs: dict[str, StoredPlaybook] = {} #: Parsed, and refused by the structural check. Kept rather than dropped #: because the reason is usually not in the file: an agent switched off, #: or one not added yet. Fixing that changes the agent table and not the @@ -75,7 +76,7 @@ def __init__( #: to be reconsidered -- a restart-shaped failure in the shape that is #: worse than a restart, since the corrective action succeeds and changes #: nothing. Re-checked on every refresh, which costs no read and no parse. - self._refused: dict[str, PlaybookSpec] = {} + self._refused: dict[str, StoredPlaybook] = {} self._index: TriggerIndex | None = None #: The agent table to validate against, asked rather than copied. It has #: to be the live one: ``apply_agents`` rebuilds the registry in place, @@ -182,7 +183,8 @@ def _refresh(self) -> None: # caller to fill is what ``load_playbook`` asks for by name, and # refusing it here would refuse the hand-written shape this refresh # exists to make visible. - if errors := validate_structure(spec, known_agents=known_agents, allow_blank_fillable=True): + errors = validation_errors(spec, known_agents) + if errors: # Refused rather than offered: a graph naming an agent that is # not on the table cannot run, and offering it spends a turn to # find that out. Said at warning level because a file the user @@ -207,7 +209,7 @@ def _refresh(self) -> None: # thing that changed is usually the agent table rather than the file. promoted = [] for name, spec in list(self._refused.items()): - if validate_structure(spec, known_agents=known_agents, allow_blank_fillable=True): + if validation_errors(spec, known_agents): continue self._specs[name] = spec del self._refused[name] @@ -285,6 +287,16 @@ def adopt(self, name: str) -> bool: self._fingerprints[name] = digest self._reindex() return False + if errors := validation_errors(spec, self._known_agents()): + logger.warning("Playbook {!r} was written but is not usable: {}", name, "; ".join(errors)) + self._specs.pop(name, None) + self._refused[name] = spec + if digest is None: + self._fingerprints.pop(name, None) + else: + self._fingerprints[name] = digest + self._reindex() + return False self._specs[name] = spec self._refused.pop(name, None) if digest is None: @@ -363,7 +375,7 @@ def listing(self, message: str = "") -> list[tuple[str, str]]: """ return self.library_view(message)[0] - def _detail(self, spec: PlaybookSpec) -> str: + def _detail(self, spec: StoredPlaybook) -> str: """One playbook as the tool description renders it.""" parts = [spec.description] if spec.params: @@ -391,7 +403,13 @@ def _detail(self, spec: PlaybookSpec) -> str: bits.append(f"one of {p.enum}") rows.append(f"{name} ({', '.join(bits)}): {p.description}") parts.append("params: " + "; ".join(rows)) - if gaps := _blank_fields(spec): + if isinstance(spec, UnifiedPlaybookSpec): + shape = ( + "composite" if spec.harness and spec.workflow else "harness-only" if spec.harness else "workflow-only" + ) + parts.append(f"v2 {shape}") + executable = spec.as_legacy_workflow() if isinstance(spec, UnifiedPlaybookSpec) else spec + if executable is not None and (gaps := _blank_fields(executable)): parts.append("left for you to fill: " + "; ".join(f"{nid}.{field}" for nid, field in gaps)) return " | ".join(parts) @@ -429,7 +447,27 @@ async def load( return None cid = self._context.get("session_key") or "" key = (cid, name) - plan = await self._executor.execute(spec, params or {}, fills=fills or {}, confirmed=confirmed) + execution_table = None + if isinstance(spec, UnifiedPlaybookSpec): + if spec.harness is not None: + from raven.agent.subagent.delegate import bind_delegate_for_turn + from raven.playbook.agent_generator import build_table + + execution_table = build_table(spec.harness, {}) + bind_delegate_for_turn(execution_table) + executable = spec.as_legacy_workflow() + if executable is None: + self._gap_rounds.pop(key, None) + return ExecutionPlan( + kind="guidance", + reply=f"Loaded Harness-only playbook '{name}'. Its workers are active for this turn; continue using spawn or run_subagent_dag.", + ) + spec = executable + + from raven.agent.subagent.delegate import delegate_scope + + with delegate_scope(execution_table): + plan = await self._executor.execute(spec, params or {}, fills=fills or {}, confirmed=confirmed) if plan.kind == "gaps": rounds = self._gap_rounds.get(key, 0) + 1 self._gap_rounds[key] = rounds @@ -448,6 +486,46 @@ async def load( self._gap_rounds.pop(key, None) return plan + def spec(self, name: str) -> StoredPlaybook | None: + """Return one offered artifact for pre-turn resolution.""" + self._refresh() + if name in self.disabled(): + return None + return self._specs.get(name) + + def harness_table(self, name: str): + """Build the durable Harness selected before a turn, if it has one.""" + spec = self.spec(name) + if not isinstance(spec, UnifiedPlaybookSpec) or spec.harness is None: + return None + from raven.playbook.agent_generator import build_table + + return build_table(spec.harness, {}) + + +def validation_errors(spec: StoredPlaybook, known_agents: list[str] | None) -> list[str]: + """Semantic findings for either on-disk Playbook contract.""" + if not isinstance(spec, UnifiedPlaybookSpec): + return validate_structure(spec, known_agents=known_agents, allow_blank_fillable=True) + if spec.state != "ready": + return ["playbook is still a draft"] + errors: list[str] = [] + if known_agents is not None and spec.harness: + errors.extend( + f"harness agent {entry.name!r} is not registered" + for entry in spec.harness.delegate + if entry.name not in known_agents + ) + executable = spec.as_legacy_workflow() + if executable is not None: + allowed = ( + [*known_agents, *(entry.label for entry in spec.harness.delegate)] + if known_agents is not None and spec.harness + else known_agents + ) + errors.extend(validate_structure(executable, known_agents=allowed, allow_blank_fillable=True)) + return errors + def _blank_fields(spec: PlaybookSpec) -> list[tuple[str, str]]: """``(node_id, field)`` for every node field the author left for the model. diff --git a/raven/playbook/store.py b/raven/playbook/store.py index 1a84b0c4f..26bf07130 100644 --- a/raven/playbook/store.py +++ b/raven/playbook/store.py @@ -35,6 +35,7 @@ from raven.agent.subagent.builtin_agents import GENERIC_AGENT from raven.playbook.types import NAME_RE, PlaybookSpec +from raven.playbook.unified import StoredPlaybook, UnifiedPlaybookSpec, is_unified_data from raven.playbook.validate import unusable_mcp_servers, validate_structure #: Playbooks that ship with the package. Kept next to the code so the @@ -71,6 +72,11 @@ def __init__(self, root: Path, *, builtin_root: Path | None = None) -> None: self._builtin_root = BUILTIN_ROOT if builtin_root is None else builtin_root self._shadow_warned: set[str] = set() + @property + def root(self) -> Path: + """Writable user-library root, also home to local Run Records.""" + return self._root + def path_for(self, name: str) -> Path: """The playbook.md that ``load`` would read: user layer first.""" user = self._root / name / "playbook.md" @@ -147,7 +153,7 @@ def _layer_ids(root: Path) -> set[str]: return set() return {p.parent.name for p in root.glob("*/playbook.md")} - def save(self, spec: PlaybookSpec, *, notes: list[str] | None = None, overwrite: bool = False) -> Path: + def save(self, spec: StoredPlaybook, *, notes: list[str] | None = None, overwrite: bool = False) -> Path: """Write one playbook into the user layer; returns the playbook.md path. ``notes`` are the generator's review lines (open questions, @@ -201,7 +207,7 @@ def save(self, spec: PlaybookSpec, *, notes: list[str] | None = None, overwrite: tmp.unlink(missing_ok=True) return path - def load(self, name: str) -> PlaybookSpec: + def load(self, name: str) -> StoredPlaybook: if self.is_shadowing(name) and name not in self._shadow_warned: # Once per store instance, not per load: the runtime re-reads the # library before every model call, so a per-load line would repeat @@ -224,13 +230,16 @@ def load(self, name: str) -> PlaybookSpec: data["description"] = front.get("description") if data["name"] != name: raise ValueError(f"playbook {name!r}: frontmatter name {data['name']!r} != directory name") - spec = PlaybookSpec.model_validate(_migrate_legacy_nodes(data, name=name)) - spec = _drop_unusable_mcp_servers(spec) + if is_unified_data(data): + spec: StoredPlaybook = _drop_unusable_mcp_servers(UnifiedPlaybookSpec.model_validate(data)) + else: + legacy = PlaybookSpec.model_validate(_migrate_legacy_nodes(data, name=name)) + spec = _drop_unusable_mcp_servers(legacy) _require_valid_structure(spec) return spec -def _drop_unusable_mcp_servers(spec: PlaybookSpec) -> PlaybookSpec: +def _drop_unusable_mcp_servers(spec: StoredPlaybook) -> StoredPlaybook: """Load a playbook without the server definitions it cannot honour. ``mcpServers`` is an optional section on top of a playbook that otherwise @@ -249,11 +258,18 @@ def _drop_unusable_mcp_servers(spec: PlaybookSpec) -> PlaybookSpec: for name, why in sorted(unusable.items()): logger.warning("playbook {}: dropping mcpServers.{} -- {}", spec.name, name, why) kept = {name: cfg for name, cfg in (spec.mcp_servers or {}).items() if name not in unusable} + if isinstance(spec, UnifiedPlaybookSpec): + if spec.workflow is None: + return spec + return spec.model_copy(update={"workflow": spec.workflow.model_copy(update={"mcp_servers": kept})}) return spec.model_copy(update={"mcp_servers": kept}) -def _require_valid_structure(spec: PlaybookSpec) -> None: - if errors := validate_structure(spec, allow_blank_fillable=True): +def _require_valid_structure(spec: StoredPlaybook) -> None: + executable = spec.as_legacy_workflow() if isinstance(spec, UnifiedPlaybookSpec) else spec + if executable is None: + return + if errors := validate_structure(executable, allow_blank_fillable=True): detail = "; ".join(errors) raise ValueError(f"playbook {spec.name!r} failed semantic validation: {detail}") @@ -342,7 +358,7 @@ def _migrate_legacy_nodes(data: dict, *, name: str) -> dict: return {**data, "nodes": migrated} -def _render(spec: PlaybookSpec, notes: list[str]) -> str: +def _render(spec: StoredPlaybook, notes: list[str]) -> str: # Serialized, not interpolated: ``load`` parses this region with # ``yaml.safe_load``, and ``description`` is model-written prose where a # colon or a leading ``#`` is ordinary. Interpolating produced files that @@ -361,13 +377,18 @@ def _render(spec: PlaybookSpec, notes: list[str]) -> str: return f"---\n{front}---\n\n{_body(spec, notes)}\n```yaml playbook-spec\n{block}```\n" -def _body(spec: PlaybookSpec, notes: list[str]) -> str: +def _body(spec: StoredPlaybook, notes: list[str]) -> str: """The human-readable region — informational only, never parsed.""" lines = [f"# {spec.name}", "", spec.description, ""] if spec.params: lines.append("Params: " + ", ".join(f"{k} ({v.description})" for k, v in spec.params.items())) - if spec.nodes: - lines.append("Steps: " + " -> ".join(n.id for n in spec.nodes)) + nodes = ( + spec.workflow.nodes if isinstance(spec, UnifiedPlaybookSpec) and spec.workflow else getattr(spec, "nodes", None) + ) + if nodes: + lines.append("Steps: " + " -> ".join(n.id for n in nodes)) + if isinstance(spec, UnifiedPlaybookSpec) and spec.harness: + lines.append("Workers: " + ", ".join(entry.label for entry in spec.harness.delegate)) if notes: lines += ["", "## Open questions", ""] lines += [f"- {note}" for note in notes] diff --git a/raven/playbook/unified.py b/raven/playbook/unified.py new file mode 100644 index 000000000..95c053424 --- /dev/null +++ b/raven/playbook/unified.py @@ -0,0 +1,200 @@ +"""The durable Playbook v2 contract: optional Harness plus optional Workflow. + +Version one files remain the workflow-only ``PlaybookSpec`` contract. Version +two is deliberately separate: the discriminator stays unambiguous on disk and +old readers fail loudly instead of treating a new artifact as a partial v1. +""" + +from __future__ import annotations + +from datetime import UTC, datetime +from typing import Any, Literal + +from pydantic import Field, model_validator + +from raven.agent.subagent.dag_graph import DagNodeSpec +from raven.config.schema import MCPServerConfig +from raven.playbook.agent_spec import AgentPlaybookSpec +from raven.playbook.types import CamelBase, ParamSpec, PlaybookSpec, Triggers + +UNIFIED_SPEC_VERSION = 2 + + +class PlaybookMatch(CamelBase): + """The portable retrieval description; ranking state stays local.""" + + summary: str = Field(min_length=1, max_length=500) + keywords: list[str] = Field(min_length=1) + + def triggers(self) -> Triggers: + return Triggers(keywords=self.keywords) + + +class InputSchema(CamelBase): + """Runtime values accepted by a reusable workflow.""" + + type: Literal["object"] = "object" + properties: dict[str, ParamSpec] = Field(default_factory=dict) + required: list[str] = Field(default_factory=list) + additional_properties: bool = False + + @model_validator(mode="after") + def _required_are_declared(self) -> "InputSchema": + missing = sorted(set(self.required) - set(self.properties)) + if missing: + raise ValueError(f"required input(s) are not declared in properties: {', '.join(missing)}") + for name, param in self.properties.items(): + if param.required != (name in self.required): + raise ValueError(f"input {name!r}: ParamSpec.required and inputSchema.required disagree") + return self + + +class WorkflowSpec(CamelBase): + """A validated, replayable DAG. No prompt-mode branch exists in v2.""" + + summary: str = Field(min_length=1) + confirm: bool = True + nodes: list[DagNodeSpec] = Field(min_length=1) + mcp_servers: dict[str, MCPServerConfig] = Field(default_factory=dict) + + +class PlaybookMetadata(CamelBase): + source_run_id: str | None = None + created_at: str = Field(default_factory=lambda: datetime.now(UTC).isoformat()) + updated_at: str = Field(default_factory=lambda: datetime.now(UTC).isoformat()) + + +class UnifiedPlaybookSpec(CamelBase): + """One reusable artifact with two independent, optional dimensions.""" + + schema_version: Literal[2] = UNIFIED_SPEC_VERSION + name: str = Field(pattern=r"^[a-z0-9][a-z0-9-]*$") + description: str = Field(min_length=1, max_length=200) + state: Literal["draft", "ready"] = "ready" + match: PlaybookMatch + input_schema: InputSchema = Field(default_factory=InputSchema) + harness: AgentPlaybookSpec | None = None + workflow: WorkflowSpec | None = None + metadata: PlaybookMetadata = Field(default_factory=PlaybookMetadata) + + @model_validator(mode="after") + def _has_an_artifact(self) -> "UnifiedPlaybookSpec": + if self.harness is None and self.workflow is None: + raise ValueError("a playbook must contain a harness, a workflow, or both") + if self.harness is not None and not self.harness.delegate: + raise ValueError("a durable harness must contain at least one worker") + return self + + @property + def triggers(self) -> Triggers: + return self.match.triggers() + + @property + def params(self) -> dict[str, ParamSpec]: + return self.input_schema.properties + + # Compatibility projection for the CLI, RPC layer and MCP helpers. These + # readers predate the unified envelope but consume only workflow fields; + # keeping that narrow face here lets old and new artifacts share those + # proven paths without teaching every helper about both disk contracts. + @property + def version(self) -> int: + return self.schema_version + + @property + def mode(self) -> Literal["dag"]: + return "dag" + + @property + def task_summary(self) -> str: + return self.workflow.summary if self.workflow else self.description + + @property + def confirm(self) -> bool: + return self.workflow.confirm if self.workflow else False + + @property + def nodes(self) -> list[DagNodeSpec] | None: + return self.workflow.nodes if self.workflow else None + + @property + def prompts(self) -> None: + return None + + @property + def mcp_servers(self) -> dict[str, MCPServerConfig]: + return self.workflow.mcp_servers if self.workflow else {} + + def as_legacy_workflow(self) -> PlaybookSpec | None: + """Project the v2 Workflow onto the proven v1 DAG executor.""" + if self.workflow is None: + return None + return PlaybookSpec( + name=self.name, + description=self.description, + task_summary=self.workflow.summary, + version=1, + mode="dag", + confirm=self.workflow.confirm, + triggers=self.triggers, + params=self.params, + mcp_servers=self.workflow.mcp_servers, + nodes=self.workflow.nodes, + ) + + def block_dump(self) -> dict[str, Any]: + data = self.model_dump(by_alias=True, exclude_none=True) + if self.harness is not None: + # A generated Harness is validated from only the fields it asked + # for. Re-emitting nested model defaults turns them into explicit + # user choices on reload; notably Checks.impl="default" then + # looks like use of the disabled checksImpl generation surface. + # Preserve explicit empty values such as tools=[] while leaving + # implicit defaults implicit. + data["harness"] = self.harness.model_dump( + by_alias=True, + exclude_none=True, + exclude_unset=True, + ) + data.pop("name", None) + data.pop("description", None) + return data + + +StoredPlaybook = PlaybookSpec | UnifiedPlaybookSpec + + +def is_unified_data(data: dict[str, Any]) -> bool: + return data.get("schemaVersion") == UNIFIED_SPEC_VERSION + + +def unified_from_legacy(spec: PlaybookSpec, *, name: str | None = None) -> UnifiedPlaybookSpec: + """Wrap one generated legacy DAG in the durable unified contract.""" + if spec.mode != "dag" or not spec.nodes: + raise ValueError("a unified Workflow must be generated as a concrete DAG") + required = [key for key, value in spec.params.items() if value.required] + return UnifiedPlaybookSpec( + name=name or spec.name, + description=spec.description, + match=PlaybookMatch(summary=spec.description, keywords=spec.triggers.keywords), + input_schema=InputSchema(properties=spec.params, required=required), + workflow=WorkflowSpec( + summary=spec.task_summary, + confirm=spec.confirm, + nodes=spec.nodes, + mcp_servers=spec.mcp_servers, + ), + ) + + +__all__ = [ + "InputSchema", + "PlaybookMatch", + "PlaybookMetadata", + "StoredPlaybook", + "UNIFIED_SPEC_VERSION", + "UnifiedPlaybookSpec", + "WorkflowSpec", + "is_unified_data", + "unified_from_legacy", +] diff --git a/raven/playbook/workflow_compiler.py b/raven/playbook/workflow_compiler.py new file mode 100644 index 000000000..9275f174e --- /dev/null +++ b/raven/playbook/workflow_compiler.py @@ -0,0 +1,249 @@ +"""Compile an accepted run DAG into a durable, parameterized v2 Workflow.""" + +from __future__ import annotations + +import json +import re +from typing import TYPE_CHECKING, Any + +from loguru import logger +from pydantic import ValidationError + +from raven.playbook.agent_spec import AgentPlaybookSpec +from raven.playbook.prompt import _inline_local_refs +from raven.playbook.types import ParamSpec, slugify +from raven.playbook.unified import ( + InputSchema, + PlaybookMatch, + PlaybookMetadata, + UnifiedPlaybookSpec, + WorkflowSpec, +) + +if TYPE_CHECKING: + from raven.agent.subagent.dag_graph import SubAgentDagSpec + from raven.providers.base import LLMProvider + +EMIT_WORKFLOW = "emit_reusable_workflow" + +SYSTEM_PROMPT = """\ +You compile one accepted execution DAG into a reusable Playbook Workflow. +Preserve its nodes, dependencies, worker aliases, skills, MCP names, inputs and +instance continuity. Replace only concrete values inside promptTemplate that +are expected to change between runs with ${params.} references and declare those inputs. +Every declared input must carry the replaced concrete value as its default so +the accepted prompt can be reconstructed exactly. Do not invent extra steps +and do not emit a prompt-mode template. Retrieval keywords +describe the user's domain and action, never the words playbook/workflow. +The Harness is supplied separately and must not be rewritten here. +""" +_PARAM_REF_RE = re.compile(r"\$\{params\.([A-Za-z_][A-Za-z0-9_-]*)\}") + + +def _tool() -> list[dict[str, Any]]: + from raven.agent.subagent.dag_graph import SubAgentDagSpec + + dag = _inline_local_refs(SubAgentDagSpec.model_json_schema(by_alias=True)) + node_items = dag["properties"]["nodes"]["items"] + param = _inline_local_refs(ParamSpec.model_json_schema(by_alias=True)) + return [ + { + "type": "function", + "function": { + "name": EMIT_WORKFLOW, + "description": "Emit the reusable Workflow compiled from the accepted DAG.", + "parameters": { + "type": "object", + "properties": { + "name": {"type": "string", "pattern": "^[a-z0-9][a-z0-9-]*$"}, + "description": {"type": "string", "maxLength": 200}, + "match": { + "type": "object", + "properties": { + "summary": {"type": "string"}, + "keywords": {"type": "array", "items": {"type": "string"}, "minItems": 1}, + }, + "required": ["summary", "keywords"], + "additionalProperties": False, + }, + "inputSchema": { + "type": "object", + "properties": { + "type": {"type": "string", "enum": ["object"]}, + "properties": {"type": "object", "additionalProperties": param}, + "required": {"type": "array", "items": {"type": "string"}}, + "additionalProperties": {"type": "boolean", "enum": [False]}, + }, + "required": ["type", "properties", "required", "additionalProperties"], + "additionalProperties": False, + }, + "workflow": { + "type": "object", + "properties": { + "summary": {"type": "string"}, + "confirm": {"type": "boolean"}, + "nodes": {"type": "array", "items": node_items, "minItems": 1}, + }, + "required": ["summary", "confirm", "nodes"], + "additionalProperties": False, + }, + }, + "required": ["name", "description", "match", "inputSchema", "workflow"], + "additionalProperties": False, + }, + }, + } + ] + + +def _args(response: Any) -> dict[str, Any] | None: + for call in getattr(response, "tool_calls", None) or []: + if isinstance(call, dict): + fn = call.get("function") or {} + name, raw = fn.get("name"), fn.get("arguments") + else: + name, raw = getattr(call, "name", None), getattr(call, "arguments", None) + if name != EMIT_WORKFLOW: + continue + if isinstance(raw, dict): + return raw + try: + parsed = json.loads(raw or "{}") + except (TypeError, ValueError): + return None + return parsed if isinstance(parsed, dict) else None + return None + + +def _prompt_is_preserved(source: str, compiled: str, artifact: UnifiedPlaybookSpec) -> bool: + """Whether declared defaults restore the exact accepted instruction.""" + if source == compiled: + return True + missing = False + + def restore(match: re.Match[str]) -> str: + nonlocal missing + param = artifact.input_schema.properties.get(match.group(1)) + if param is None or param.default is None: + missing = True + return "" + return str(param.default) + + restored, references = _PARAM_REF_RE.subn(restore, compiled) + return references > 0 and not missing and restored == source + + +def _preservation_errors(accepted: "SubAgentDagSpec", artifact: UnifiedPlaybookSpec) -> list[str]: + """Ensure compilation parameterizes prompts without rewriting the proven graph.""" + if artifact.workflow is None: + return ["compiled artifact has no Workflow"] + expected_ids = [node.id for node in accepted.nodes] + actual_ids = [node.id for node in artifact.workflow.nodes] + errors: list[str] = [] + if artifact.workflow.confirm != accepted.confirm: + errors.append("workflow confirm changed during compilation") + if actual_ids != expected_ids: + errors.append(f"workflow node ids/order changed: expected {expected_ids}, got {actual_ids}") + return errors + actual = {node.id: node for node in artifact.workflow.nodes} + for source in accepted.nodes: + compiled = actual[source.id] + for field in ("subagent", "node_summary", "depends_on", "skills", "mcps", "inputs", "instance"): + if getattr(compiled, field) != getattr(source, field): + errors.append(f"node {source.id!r}: {field} changed during compilation") + if not _prompt_is_preserved(source.prompt_template, compiled.prompt_template, artifact): + errors.append(f"node {source.id!r}: prompt_template changed beyond declared parameter defaults") + return errors + + +class WorkflowCompiler: + """One compile call plus one repair, with a lossless deterministic fallback.""" + + def __init__(self, provider: "LLMProvider", model: str | None = None) -> None: + self._provider = provider + self._model = model + + async def compile( + self, + *, + query: str, + dag: "SubAgentDagSpec", + run_id: str, + harness: AgentPlaybookSpec | None = None, + name_hint: str | None = None, + description_hint: str | None = None, + ) -> UnifiedPlaybookSpec: + payload = { + "query": query, + "acceptedDag": dag.model_dump(by_alias=True, exclude_none=True), + "workerAliases": [entry.label for entry in harness.delegate] if harness else [], + "nameHint": name_hint, + "descriptionHint": description_hint, + } + messages: list[dict[str, Any]] = [ + {"role": "system", "content": SYSTEM_PROMPT}, + {"role": "user", "content": json.dumps(payload, ensure_ascii=False, indent=2)}, + ] + for _ in range(2): + try: + response = await self._provider.chat_with_retry( + messages=messages, + tools=_tool(), + model=self._model or None, + tool_choice={"type": "function", "function": {"name": EMIT_WORKFLOW}}, + ) + data = _args(response) + if data is None: + raise ValueError(f"model did not call {EMIT_WORKFLOW}") + data["schemaVersion"] = 2 + data["state"] = "ready" + data["harness"] = harness.model_dump(by_alias=True, exclude_none=True) if harness else None + data["metadata"] = PlaybookMetadata(source_run_id=run_id).model_dump(by_alias=True) + artifact = UnifiedPlaybookSpec.model_validate(data) + if errors := _preservation_errors(dag, artifact): + raise ValueError("; ".join(errors)) + from raven.playbook.runtime import validation_errors + + allowed_agents = {node.subagent for node in dag.nodes} + if harness: + allowed_agents.update(entry.name for entry in harness.delegate) + allowed_agents.update(entry.label for entry in harness.delegate) + if errors := validation_errors(artifact, sorted(allowed_agents)): + raise ValueError("; ".join(errors)) + return artifact + except (ValidationError, ValueError, TypeError) as exc: + messages.append( + { + "role": "user", + "content": f"The compiled Workflow was invalid: {exc}. Emit the complete corrected artifact.", + } + ) + except Exception as exc: # noqa: BLE001 - compilation has a safe fallback + logger.warning("playbook workflow compiler failed: {}", exc) + break + return self._fallback(query, dag, run_id, harness, name_hint, description_hint) + + @staticmethod + def _fallback( + query: str, + dag: "SubAgentDagSpec", + run_id: str, + harness: AgentPlaybookSpec | None, + name_hint: str | None, + description_hint: str | None, + ) -> UnifiedPlaybookSpec: + words = [word.lower() for word in re.findall(r"[A-Za-z0-9][A-Za-z0-9_-]{2,}", query)[:8]] + summary = (description_hint or query.strip().splitlines()[0] or dag.task_summary)[:200] + name = slugify(name_hint or dag.task_summary or summary) + return UnifiedPlaybookSpec( + name=name, + description=summary, + match=PlaybookMatch(summary=summary, keywords=words or [summary[:80]]), + input_schema=InputSchema(), + harness=harness, + workflow=WorkflowSpec(summary=dag.task_summary, confirm=dag.confirm, nodes=dag.nodes), + metadata=PlaybookMetadata(source_run_id=run_id), + ) + + +__all__ = ["EMIT_WORKFLOW", "WorkflowCompiler"] diff --git a/raven/rpc/methods/playbooks.py b/raven/rpc/methods/playbooks.py index d444d026e..b84ef0232 100644 --- a/raven/rpc/methods/playbooks.py +++ b/raven/rpc/methods/playbooks.py @@ -57,6 +57,22 @@ def _shape(spec: PlaybookSpec) -> list[dict[str, Any]]: return [{"id": node.id, "depends_on": list(node.depends_on)} for node in (spec.nodes or [])] +def _artifact_fields(spec: Any, *, detail: bool = False) -> dict[str, Any]: + """Unified-only fields beside the legacy graph projection.""" + from raven.playbook.unified import UnifiedPlaybookSpec + + if not isinstance(spec, UnifiedPlaybookSpec): + return {"schema_version": 1, "artifact_kind": "legacy", "workers": []} + kind = "composite" if spec.harness and spec.workflow else "harness" if spec.harness else "workflow" + workers = [] + for entry in spec.harness.delegate if spec.harness else []: + worker = {"label": entry.label, "agent": entry.name} + if detail: + worker["brief"] = entry.brief + workers.append(worker) + return {"schema_version": spec.schema_version, "artifact_kind": kind, "workers": workers} + + def _row(store: PlaybookStore, name: str, disabled: set[str]) -> dict[str, Any]: origin = store.origin_of(name) or "user" row: dict[str, Any] = { @@ -68,6 +84,9 @@ def _row(store: PlaybookStore, name: str, disabled: set[str]) -> dict[str, Any]: "mode": "dag", "confirm": True, "nodes": [], + "schema_version": 1, + "artifact_kind": "legacy", + "workers": [], "error": "", } try: @@ -80,6 +99,7 @@ def _row(store: PlaybookStore, name: str, disabled: set[str]) -> dict[str, Any]: row["mode"] = spec.mode row["confirm"] = spec.confirm row["nodes"] = _shape(spec) + row.update(_artifact_fields(spec)) return row @@ -133,6 +153,7 @@ async def playbooks_get(params: dict) -> dict: return { "playbook": { "name": spec.name, + **_artifact_fields(spec, detail=True), "description": spec.description, "task_summary": spec.task_summary, "version": spec.version, @@ -471,7 +492,7 @@ async def playbooks_validate( import yaml from pydantic import ValidationError - from raven.playbook.validate import validate_structure + from raven.playbook.runtime import validation_errors name = _known_name(params.get("name")) store = _store() @@ -488,7 +509,7 @@ async def playbooks_validate( except (ValidationError, ValueError, yaml.YAMLError) as exc: errors.append(str(exc)) if spec is not None: - errors.extend(validate_structure(spec, known_agents=_known_agent_names(agent_loop_factory))) + errors.extend(validation_errors(spec, _known_agent_names(agent_loop_factory))) return {"name": name, "ok": not errors, "errors": errors, "path": str(store.path_for(name))} @@ -712,7 +733,7 @@ async def playbooks_create( budget = _generation_budget_s() try: - generated = await asyncio.wait_for(runtime.generator.generate(workflow, skills), budget) + generated = await asyncio.wait_for(runtime.generator.generate(workflow, skills, dag_only=True), budget) except PlaybookGenerationError as exc: return { "name": name, @@ -732,7 +753,9 @@ async def playbooks_create( "adopted": False, } - spec = generated.spec.model_copy(update={"name": name}) + from raven.playbook.unified import unified_from_legacy + + spec = unified_from_legacy(generated.spec, name=name) try: path = store.save(spec, notes=generated.notes) except PlaybookExistsError as exc: diff --git a/raven/rpc/methods/turn.py b/raven/rpc/methods/turn.py index e80f07ecf..f1b65793d 100644 --- a/raven/rpc/methods/turn.py +++ b/raven/rpc/methods/turn.py @@ -441,6 +441,7 @@ async def turn_send( surface=declared_surface(), ), text=parsed.content, + playbook_mode=parsed.playbook_mode, media=_resolve_media(parsed.media), # conversation == the lane. For the main agent that is the session key, # which is also the front-end subscription key; for a direct chat it is diff --git a/raven/rpc/models.py b/raven/rpc/models.py index 90471fee8..0c9fc2f0d 100644 --- a/raven/rpc/models.py +++ b/raven/rpc/models.py @@ -1289,6 +1289,8 @@ class SessionHistoryResult(_Strict): class TurnSendParams(_Strict): session_key: str content: str + playbook_mode: Literal["off", "task", "persona"] | None = None + """Optional per-turn override for dynamic Playbook generation.""" channel: str | None = None chat_id: str | None = None sender_id: str | None = None @@ -3930,6 +3932,19 @@ class PlaybookNodeShape(_Strict): depends_on: list[str] +class PlaybookWorkerShape(_Strict): + """One durable Harness alias and the registered agent behind it.""" + + label: str + agent: str + + +class PlaybookWorker(PlaybookWorkerShape): + """The full worker detail; its brief is the durable per-job instruction.""" + + brief: str + + class PlaybookRow(_Strict): """One playbook as the library list needs it. @@ -3942,6 +3957,9 @@ class PlaybookRow(_Strict): name: str description: str task_summary: str + schema_version: int + artifact_kind: Literal["legacy", "workflow", "harness", "composite"] + workers: list[PlaybookWorkerShape] mode: Literal["dag", "prompt"] confirm: bool origin: str @@ -4026,6 +4044,9 @@ class PlaybookDetail(_Strict): description: str task_summary: str version: int + schema_version: int + artifact_kind: Literal["legacy", "workflow", "harness", "composite"] + workers: list[PlaybookWorker] mode: Literal["dag", "prompt"] confirm: bool origin: str @@ -4692,6 +4713,8 @@ class SubagentCancelInstanceResult(_Strict): "PlaybookDetail", "PlaybookNode", "PlaybookNodeShape", + "PlaybookWorker", + "PlaybookWorkerShape", "PlaybookParam", "PlaybookRow", "PlaybooksGetParams", diff --git a/raven/spine/turn.py b/raven/spine/turn.py index d0b54a587..21859dfb7 100644 --- a/raven/spine/turn.py +++ b/raven/spine/turn.py @@ -80,6 +80,9 @@ class TurnRequest: # transcript -- a direct chat exists to keep those exchanges out of the main # agent's context. See AgentLoop.run_turn. direct_target: tuple[str, str] | None = None + # Per-turn UI choice for Playbook generation. None inherits the configured + # default; off/task/persona are resolved before the setup model call. + playbook_mode: str | None = None # A lane is a serial domain, so an instance that is to answer while the main diff --git a/rpc-schema/openrpc.json b/rpc-schema/openrpc.json index b574dd3f6..5053a5799 100644 --- a/rpc-schema/openrpc.json +++ b/rpc-schema/openrpc.json @@ -596,6 +596,18 @@ "type": "string" } }, + { + "name": "playbook_mode", + "required": false, + "schema": { + "type": "string", + "enum": [ + "off", + "task", + "persona" + ] + } + }, { "name": "channel", "required": false, @@ -11919,6 +11931,44 @@ "type": "object", "description": "Just enough of one step to draw the graph: which step it is, and what it\nwaits for. The library page draws a concept diagram per card, and shipping\nthe prompts and per-node config the detail view needs would be the whole\nlibrary on page open." }, + "PlaybookWorkerShape": { + "additionalProperties": false, + "properties": { + "label": { + "type": "string" + }, + "agent": { + "type": "string" + } + }, + "required": [ + "label", + "agent" + ], + "type": "object", + "description": "One durable Harness alias and the registered agent behind it." + }, + "PlaybookWorker": { + "additionalProperties": false, + "properties": { + "label": { + "type": "string" + }, + "agent": { + "type": "string" + }, + "brief": { + "type": "string" + } + }, + "required": [ + "label", + "agent", + "brief" + ], + "type": "object", + "description": "The full worker detail; its brief is the durable per-job instruction." + }, "PlaybookRow": { "additionalProperties": false, "properties": { @@ -11931,6 +11981,24 @@ "task_summary": { "type": "string" }, + "schema_version": { + "type": "integer" + }, + "artifact_kind": { + "enum": [ + "legacy", + "workflow", + "harness", + "composite" + ], + "type": "string" + }, + "workers": { + "items": { + "$ref": "#/components/schemas/PlaybookWorkerShape" + }, + "type": "array" + }, "mode": { "enum": [ "dag", @@ -11961,6 +12029,9 @@ "name", "description", "task_summary", + "schema_version", + "artifact_kind", + "workers", "mode", "confirm", "origin", @@ -12071,6 +12142,24 @@ "version": { "type": "integer" }, + "schema_version": { + "type": "integer" + }, + "artifact_kind": { + "enum": [ + "legacy", + "workflow", + "harness", + "composite" + ], + "type": "string" + }, + "workers": { + "items": { + "$ref": "#/components/schemas/PlaybookWorker" + }, + "type": "array" + }, "mode": { "enum": [ "dag", @@ -12124,6 +12213,9 @@ "task_summary", "version", "mode", + "schema_version", + "artifact_kind", + "workers", "confirm", "origin", "disabled", diff --git a/tests/integration/test_agent_harness_real_llm.py b/tests/integration/test_agent_harness_real_llm.py new file mode 100644 index 000000000..19ebbfa1b --- /dev/null +++ b/tests/integration/test_agent_harness_real_llm.py @@ -0,0 +1,315 @@ +"""Live two-model-call E2E for generated worker harnesses.""" + +from __future__ import annotations + +import json +import os +from pathlib import Path + +import pytest + +from raven.agent.subagent.charter import parse +from raven.playbook.agent_generator import WorkerTableGenerator +from raven.providers.litellm_provider import LiteLLMProvider + + +def _config_key() -> str: + path = Path.home() / ".raven" / "config.json" + try: + section = json.loads(path.read_text(encoding="utf-8")).get("providers", {}).get("openrouter") or {} + except Exception: + return "" + return section.get("apiKey") or section.get("api_key") or "" + + +OPENROUTER_KEY = os.environ.get("OPENROUTER_API_KEY") or _config_key() +MODEL = "openrouter/anthropic/claude-fable-5" +MARKER = "HARNESS_E2E_OK" + +pytestmark = [ + pytest.mark.real_llm, + pytest.mark.slow, + pytest.mark.skipif(not OPENROUTER_KEY, reason="no OpenRouter credential (env or ~/.raven)"), +] + + +@pytest.mark.asyncio +async def test_generated_harness_reaches_a_real_worker_model() -> None: + provider = LiteLLMProvider(api_key=OPENROUTER_KEY, default_model=MODEL, provider_name="openrouter") + table = await WorkerTableGenerator(provider, model=MODEL).generate( + ( + "Delegate exactly one Raven-Research worker. Its job is to return the exact marker " + f"{MARKER}. Give it a system prompt requiring that exact marker, allow only web_search, " + "and set its completion condition to returning the marker." + ), + ["Raven-Research"], + ["web_search", "web_fetch"], + {"Raven-Research": "researches a bounded question and reports the answer"}, + ) + + assert table is not None + worker = table.get("Raven-Research") + assert worker is not None and worker.payload is not None + assert worker.payload.get("tools") == ["web_search"] + assert MARKER in worker.payload.get("instructionAddendum", "") + assert MARKER in worker.payload.get("stopWhen", "") + + charter = parse(worker.payload) + assert charter is not None + response = await provider.chat_with_retry( + messages=[ + {"role": "system", "content": charter.task_brief}, + {"role": "user", "content": "Return the required marker now."}, + ], + model=MODEL, + ) + assert MARKER in (response.content or "") + + +@pytest.mark.asyncio +async def test_real_model_generated_checks_and_judge_execute(tmp_path) -> None: + from raven.agent.loop import AgentLoop + from raven.agent.loop.bundles import ToolWiring + from raven.agent.subagent.charter import charter_scope, judge + + provider = LiteLLMProvider(api_key=OPENROUTER_KEY, default_model=MODEL, provider_name="openrouter") + table = await WorkerTableGenerator(provider, model=MODEL).generate( + ( + "Delegate exactly one Raven-Code worker. Allow only write_file. Add a declarative check requiring " + "write_file paths to start with out/. Also write a Python judge that refuses write_file when its " + "content contains SECRET, returning the sentence 'secret content forbidden'." + ), + ["Raven-Code"], + ["read_file", "write_file"], + {"Raven-Code": "edits local code and files within explicit boundaries"}, + ) + + assert table is not None + worker = table.get("Raven-Code") + assert worker is not None and worker.payload is not None + assert worker.payload.get("checks") + assert worker.payload.get("code") + charter = parse(worker.payload) + assert charter is not None + with charter_scope(charter): + assert judge("write_file", {"path": "elsewhere/a.txt", "content": "safe"}, ()) + assert judge("write_file", {"path": "out/a.txt", "content": "SECRET"}, ()) == ["secret content forbidden"] + assert judge("write_file", {"path": "out/a.txt", "content": "safe"}, ()) == [] + + loop = AgentLoop(provider=provider, workspace=tmp_path, model=MODEL, tools=ToolWiring(restrict_to_workspace=True)) + (tmp_path / "out").mkdir() + with charter_scope(charter): + refused = await loop.tools.execute("write_file", {"path": "out/blocked.txt", "content": "SECRET"}) + assert "secret content forbidden" in str(refused) + assert not (tmp_path / "out" / "blocked.txt").exists() + + +@pytest.mark.asyncio +async def test_real_model_cannot_emit_disabled_harness_fields(tmp_path, monkeypatch) -> None: + from raven.agent import harness_capabilities + + document = json.loads(harness_capabilities._PATH.read_text(encoding="utf-8")) + root = document["harnessGeneration"] + for module, name in ( + ("memory", "systemPrompt"), + ("memory", "stopWhen"), + ("capability", "tools"), + ("action", "checks"), + ): + root[module]["parameters"]["items"][name]["enabled"] = False + for module, name in (("memory", "intake"), ("planning", "advise"), ("action", "judge"), ("action", "salvage")): + root[module]["functions"]["participant"]["items"][name]["enabled"] = False + path = tmp_path / "harness_generation.json" + path.write_text(json.dumps(document), encoding="utf-8") + monkeypatch.setattr(harness_capabilities, "_PATH", path) + + provider = LiteLLMProvider(api_key=OPENROUTER_KEY, default_model=MODEL, provider_name="openrouter") + table = await WorkerTableGenerator(provider, model=MODEL).generate( + "Delegate exactly one Raven-Research worker to return a short answer about the task.", + ["Raven-Research"], + ["web_search", "web_fetch"], + {"Raven-Research": "researches a bounded question and reports the answer"}, + ) + + assert table is not None + worker = table.get("Raven-Research") + assert worker is not None + assert set(worker.payload or {}) <= {"brief", "prompt", "timeoutSeconds"} + + +@pytest.mark.asyncio +async def test_real_harness_crosses_acp_process_and_blocks_worker_tool(tmp_path) -> None: + import sys + + from raven.acp_client.acp_agent import AcpAgentBackend + from raven.acp_client.pool import close_pool + from raven.agent.subagent.delegate import dispatch_charter + + provider = LiteLLMProvider(api_key=OPENROUTER_KEY, default_model=MODEL, provider_name="openrouter") + table = await WorkerTableGenerator(provider, model=MODEL).generate( + ( + "Delegate exactly one Raven-ACP worker. Allow only write_file. Give it instructions to attempt writing " + "out/blocked.txt with content SECRET, then return HARNESS_ACP_BLOCKED after the refusal. Add a Python " + "judge that refuses write_file when content contains SECRET with reason 'secret content forbidden'." + ), + ["Raven-ACP"], + ["write_file"], + {"Raven-ACP": "runs a Raven agent loop in a separate ACP process"}, + ) + assert table is not None + worker = table.get("Raven-ACP") + assert worker is not None and worker.payload is not None + assert worker.payload.get("code") + + workspace = tmp_path / "workspace" + workspace.mkdir() + home = tmp_path / "worker-home" + home.mkdir() + (home / "config.json").write_text( + json.dumps( + { + "providers": {"openrouter": {"apiKey": OPENROUTER_KEY}}, + "agents": { + "defaults": { + "provider": "openrouter", + "model": MODEL, + "maxToolIterations": 4, + "requestTimeoutSeconds": 120, + } + }, + "memory": {"backend": None}, + "playbooks": {"enabled": False}, + "sessionTitle": {"enabled": False}, + "tools": {"restrictToWorkspace": True}, + } + ), + encoding="utf-8", + ) + binary = Path(sys.executable).with_name("raven.exe" if sys.platform == "win32" else "raven") + assert binary.exists() + backend = AcpAgentBackend( + name=f"harness-e2e-{tmp_path.name}", + command=f"{binary} acp", + env={"RAVEN_HOME": str(home)}, + ready_timeout_ms=120_000, + timeout=180, + ) + try: + with dispatch_charter(worker.payload): + reply = await backend.run( + "Follow the dispatch brief and return its required marker after the tool result.", + task_id="harness-live-e2e", + workspace=workspace, + executor=None, + session_key="test:harness-live-e2e", + provider=provider, + model=MODEL, + ) + finally: + await close_pool() + + assert "HARNESS_ACP_BLOCKED" in reply + frames = "\n".join(path.read_text(encoding="utf-8") for path in (tmp_path / "acp-frames").rglob("*.jsonl")) + assert "write_file" in frames and "out/blocked.txt" in frames + assert "secret content forbidden" in frames + assert '"status": "failed"' in frames + assert not (workspace / "out" / "blocked.txt").exists() + + +@pytest.mark.asyncio +async def test_real_model_generates_binds_and_executes_three_participant_functions() -> None: + from raven.agent.harness.participants import compose_advice, compose_intake, compose_salvage + from raven.agent.subagent.charter import charter_participants, charter_scope + from raven.contracts.participant import StepView + + provider = LiteLLMProvider(api_key=OPENROUTER_KEY, default_model=MODEL, provider_name="openrouter") + table = await WorkerTableGenerator(provider, model=MODEL).generate( + ( + "Delegate exactly one Raven-Code worker and emit all three optional generated functions. " + "Under functions, write Python source for intake(text, step) returning " + "{'text': 'INTAKE_OK'}, advise(step) returning exactly 'ADVISE_OK', and salvage(step) " + "returning exactly 'SALVAGE_OK'. Keep every function to a single return statement." + ), + ["Raven-Code"], + ["read_file"], + {"Raven-Code": "edits local code and files"}, + ) + + assert table is not None + worker = table.get("Raven-Code") + assert worker is not None and worker.payload is not None + assert set(worker.payload.get("functions", {})) == {"intake", "advise", "salvage"} + charter = parse(worker.payload) + assert charter is not None + step = StepView( + session_key="live", + iteration=1, + response=None, + transcript=(), + history=(), + turn_base=0, + question="test generated functions", + rollbacks=0, + mode=None, + mode_overlay=None, + phase="user_inbound", + ) + with charter_scope(charter): + participants = charter_participants() + intake = await compose_intake("ignored", step, participants) + assert intake is not None and intake.text == "INTAKE_OK" + assert await compose_advice(step, participants) == "ADVISE_OK" + assert await compose_salvage(step, participants) == "SALVAGE_OK" + + +@pytest.mark.asyncio +async def test_generated_intake_executes_inside_the_forked_acp_worker(tmp_path) -> None: + import sys + + from raven.acp_client.acp_agent import AcpAgentBackend + from raven.acp_client.pool import close_pool + from raven.agent.subagent.delegate import dispatch_charter + + marker = "FORKED_INTAKE_SHORT_CIRCUIT_OK" + payload = { + "functions": { + "intake": (f"def intake(text, step):\n return {{'reply': '{marker}', 'note': step.get('phase')}}") + } + } + workspace = tmp_path / "workspace" + workspace.mkdir() + home = tmp_path / "worker-home" + home.mkdir() + (home / "config.json").write_text( + json.dumps( + { + "providers": {"openrouter": {"apiKey": OPENROUTER_KEY}}, + "agents": {"defaults": {"provider": "openrouter", "model": MODEL}}, + "memory": {"backend": None}, + "playbooks": {"enabled": False}, + "sessionTitle": {"enabled": False}, + } + ), + encoding="utf-8", + ) + binary = Path(sys.executable).with_name("raven.exe" if sys.platform == "win32" else "raven") + backend = AcpAgentBackend( + name=f"harness-functions-{tmp_path.name}", + command=f"{binary} acp", + env={"RAVEN_HOME": str(home)}, + ready_timeout_ms=120_000, + timeout=120, + ) + try: + with dispatch_charter(payload): + reply = await backend.run( + "This text must never reach the worker model.", + task_id="harness-function-acp", + workspace=workspace, + executor=None, + session_key="test:harness-function-acp", + ) + finally: + await close_pool() + + assert reply == marker diff --git a/tests/integration/test_agent_playbook_e2e.py b/tests/integration/test_agent_playbook_e2e.py index 883d581fe..22fea4f4e 100644 --- a/tests/integration/test_agent_playbook_e2e.py +++ b/tests/integration/test_agent_playbook_e2e.py @@ -24,24 +24,25 @@ from raven.agent.loop.bundles import EngineWiring, ToolWiring, TurnPolicy from raven.config.raven import CheckpointConfig, RuntimeConfig from raven.config.schema import PlaybookConfig +from raven.playbook.agent_generator import TASK_TOOL from raven.providers.base import LLMProvider, LLMResponse, ToolCallRequest from raven.spine.message import ChatType, Source from raven.spine.turn import Origin, TurnRequest QUERY = "List what is in the workspace, then tell me how many entries you saw." -EMIT = "emit_worker_table" +EMIT = TASK_TOOL WORKERS = { + "artifactName": "competitor-research", "description": "two researchers, one per competitor", "workers": [ { "as": "research-a", - "name": "Raven", - "brief": "only A's pricing", - "systemPrompt": "Only look at A. Leave B alone.", - "tools": ["web_fetch"], + "agent": "Raven", + "prompt": "Only research A's pricing and leave B alone", + "tools": ["list_dir"], }, - {"as": "research-b", "name": "Raven", "brief": "only B's pricing", "systemPrompt": "Only look at B."}, + {"as": "research-b", "agent": "Raven", "prompt": "Only research B's pricing and leave A alone"}, ], } @@ -146,7 +147,10 @@ async def _run(workspace, harness: str, *, emit_table: bool = True) -> tuple[_Sc tools=ToolWiring(restrict_to_workspace=True), engine=EngineWiring( runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), - playbook_config=PlaybookConfig(agentHarness=harness), + playbook_config=PlaybookConfig( + dir=str(workspace / "playbooks"), + agentHarness=harness, + ), ), ) emitted: list[str] = [] @@ -190,9 +194,9 @@ async def test_the_default_turn_is_deterministic(tmp_path_factory) -> None: async def test_a_configured_turn_differs_only_in_what_spawn_offers(tmp_path_factory) -> None: """The switch's promise, measured rather than asserted. - Everything the loop hands the provider -- the model, the tool array, every - message -- is identical with the feature on. The single exception is the - target ``spawn`` offers, which is the whole of what a worker table is for. + The setup does not add or remove any main-turn model call and does not + change the answer. Its generated Harness is allowed to change dispatch + targets and the live Playbook listing because it is saved immediately. """ off, off_emitted = await _run(tmp_path_factory.mktemp("off"), "default") on, on_emitted = await _run(tmp_path_factory.mktemp("on"), "generate") @@ -203,8 +207,7 @@ async def test_a_configured_turn_differs_only_in_what_spawn_offers(tmp_path_fact for index, (a, b) in enumerate(zip(off_turns, on_turns, strict=True), 1): assert a["model"] == b["model"], f"call {index}: model moved" - assert a["tool_names"] == b["tool_names"], f"call {index}: the tool array moved" - assert a["messages"] == b["messages"], f"call {index}: the prompt moved" + assert set(a["tool_names"]) == set(b["tool_names"]), f"call {index}: the tool surface moved" @pytest.mark.asyncio @@ -220,8 +223,8 @@ async def test_spawn_offers_the_workers_and_their_briefs(tmp_path_factory) -> No on, _ = await _run(tmp_path_factory.mktemp("on"), "generate") target = _turn_calls(on)[0]["spawn_target"] assert target["enum"] == ["research-a", "research-b"] - assert "only A's pricing" in target["description"] - assert "only B's pricing" in target["description"] + assert "Only research A's pricing" in target["description"] + assert "Only research B's pricing" in target["description"] assert "Raven" in target["description"], "the label says which agent is behind it" @@ -260,7 +263,7 @@ async def test_the_setup_call_runs_on_the_session_s_own_model(tmp_path_factory) tools=ToolWiring(restrict_to_workspace=True), engine=EngineWiring( runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), - playbook_config=PlaybookConfig(agentHarness="generate"), + playbook_config=PlaybookConfig(dir=str(workspace / "playbooks"), agentHarness="generate"), ), ) switched = replace(loop.binding_for_session("test:c1"), model="switched-model") diff --git a/tests/integration/test_subagent_charter_e2e.py b/tests/integration/test_subagent_charter_e2e.py index 28b1523c7..965edb111 100644 --- a/tests/integration/test_subagent_charter_e2e.py +++ b/tests/integration/test_subagent_charter_e2e.py @@ -85,7 +85,9 @@ def _spec(**over) -> AgentPlaybookSpec: def test_the_generated_table_carries_a_payload_the_worker_can_read() -> None: worker = build_table(_spec(), {"research-a": "only A's pricing"}).get("research-a") - assert worker.payload["prompt"] == BRIEF + assert worker.payload["brief"] == "only A's pricing" + assert worker.payload["instructionAddendum"] == BRIEF + assert worker.payload["prompt"] == "only A's pricing\n\n" + BRIEF assert worker.payload["tools"] == ["web_search", "web_fetch"] assert worker.payload["stopWhen"] == "both tables land" @@ -113,7 +115,9 @@ def test_the_payload_survives_the_round_trip_to_a_charter() -> None: worker = build_table(_spec(), {"research-a": "b"}).get("research-a") charter = parse(worker.payload) assert charter is not None - assert charter.prompt == BRIEF + assert charter.prompt == "b" + assert charter.instruction_addendum == BRIEF + assert charter.task_brief == "b\n\n" + BRIEF assert charter.tools == ("web_search", "web_fetch") assert charter.stop_when == "both tables land" @@ -596,7 +600,9 @@ def test_every_field_a_playbook_can_write_reaches_the_worker() -> None: assert charter is not None with charter_scope(charter): - assert charter.prompt == BRIEF + assert charter.prompt == "b" + assert charter.instruction_addendum == BRIEF + assert charter.task_brief == "b\n\n" + BRIEF assert charter.stop_when == "both tables land" assert narrowed_tools(["grep", "write_file", "exec"]) == frozenset({"exec"}) assert judge("write_file", {"path": "elsewhere", "content": "x"}, ()) == ["under out/ only"] @@ -612,29 +618,79 @@ def test_nothing_the_model_may_emit_is_quietly_dropped() -> None: The failure this catches is silent by construction: a field offered to the generating model and consumed by nobody produces a table that validates, a dispatch that runs, and a brief that is missing half of what its author - wrote -- with no error anywhere. ``code`` and ``timeoutSeconds`` were in - exactly that state. + wrote -- with no error anywhere. Judge's external ``functions`` entry and + ``timeoutSeconds`` were once in exactly that state. """ - from raven.playbook.agent_generator import _spec_from_args, emit_tool + from raven.playbook.agent_generator import _persona_spec_from_args, persona_tool - schema = emit_tool(["Raven"], ["grep", "write_file"])[0]["function"]["parameters"] + schema = persona_tool(["Raven"], ["grep", "write_file"])[0]["function"]["parameters"] offered = set(schema["properties"]["workers"]["items"]["properties"]) row: dict = { - "name": "Raven", + "agent": "Raven", "as": "w", "brief": "a brief", "systemPrompt": BRIEF, "stopWhen": "it lands", "tools": ["grep"], "checks": [{"tool": "write_file", "pathPrefix": "out/"}], - "code": "def judge(name, params, prior):\n return []\n", + "functions": { + "intake": "def intake(text, step):\n return {'text': text}", + "advise": "def advise(step):\n return None", + "judge": "def judge(name, params, prior):\n return []\n", + "salvage": "def salvage(step):\n return None", + }, "timeoutSeconds": 90, } assert offered == set(row), "this test must exercise exactly what the schema offers" - spec, briefs = _spec_from_args({"workers": [row]}, {"Raven"}) + spec, briefs = _persona_spec_from_args({"workers": [row]}, {"Raven"}) payload = build_table(spec, briefs).get("w").payload assert briefs["w"] == "a brief" - assert set(payload) == {"prompt", "tools", "stopWhen", "checks", "code", "timeoutSeconds"} + assert set(payload) == { + "brief", + "instructionAddendum", + "prompt", + "tools", + "stopWhen", + "checks", + "code", + "functions", + "timeoutSeconds", + } + + +@pytest.mark.asyncio +async def test_generated_functions_execute_in_the_in_process_worker_loop(workspace) -> None: + from raven.agent.subagent.backends.raven_loop import RavenLoopBackend + + class _EmptyCapture(_Stub): + def __init__(self) -> None: + self.calls: list[list[dict]] = [] + + async def chat(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + self.calls.append([dict(message) for message in messages]) + return LLMResponse(content=None, finish_reason="stop") + + async def chat_with_retry(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self.chat(messages, tools=tools, model=model, **kwargs) + + provider = _EmptyCapture() + backend = RavenLoopBackend(provider=provider, model="stub", agent_home=workspace) + payload = { + "functions": { + "intake": "def intake(text, step):\n return {'text': 'INTAKE_RUNTIME_OK'}", + "advise": "def advise(step):\n return 'ADVISE_RUNTIME_OK'", + "salvage": "def salvage(step):\n return 'SALVAGE_RUNTIME_OK'", + } + } + + with dispatch_charter(payload): + result = await backend.run("original task", task_id="generated-functions", workspace=workspace, executor=None) + + assert result == "SALVAGE_RUNTIME_OK" + first_prompt = "\n".join(str(message.get("content", "")) for message in provider.calls[0]) + assert "INTAKE_RUNTIME_OK" in first_prompt + assert "original task" not in first_prompt + assert "ADVISE_RUNTIME_OK" in first_prompt diff --git a/tests/integration/test_unified_playbook_real_llm.py b/tests/integration/test_unified_playbook_real_llm.py new file mode 100644 index 000000000..651aebbd6 --- /dev/null +++ b/tests/integration/test_unified_playbook_real_llm.py @@ -0,0 +1,358 @@ +"""Real-model E2Es for unified Playbook decisions and persistence. + +These tests use the current Codex OAuth session through a permission-restricted +temporary adapter. They never print or persist credentials in the repository. +Run explicitly: + + uv run pytest tests/integration/test_unified_playbook_real_llm.py -v -s +""" + +from __future__ import annotations + +import json +import os +from pathlib import Path + +import pytest + +from raven.agent.loop import AgentLoop +from raven.agent.loop.bundles import EngineWiring, SubagentWiring, ToolWiring, TurnPolicy +from raven.agent.subagent.dag_graph import DagNodeSpec, SubAgentDagSpec +from raven.agent.tools.load_playbook import LoadPlaybookTool +from raven.config.raven import CheckpointConfig, RuntimeConfig +from raven.config.schema import BuiltinAgentConfig, PlaybookConfig +from raven.playbook.agent_generator import PersonaPlaybookGenerator, TaskPlaybookGenerator +from raven.playbook.agent_spec import AgentPlaybookSpec, DelegateEntry +from raven.playbook.store import PlaybookStore +from raven.playbook.unified import UnifiedPlaybookSpec +from raven.playbook.workflow_compiler import WorkflowCompiler +from raven.providers.openai_codex_provider import OpenAICodexProvider +from raven.spine.message import ChatType, Source +from raven.spine.turn import Origin, TurnRequest + +MODEL = "openai-codex/gpt-5.6-sol" +_CODEX_AUTH = Path(os.environ.get("CODEX_HOME", str(Path.home() / ".codex"))) / "auth.json" +TRAVEL_PERSONA_PROMPT = ( + "我每年会独立旅行几次,请为我创建并保存一个名为 travel-concierge 的可复用旅游助手。" + "以后我只想提供目的地、日期、总预算、同行人和旅行节奏偏好,它就能给出真正可以照着走的方案。" + "它需要调查最新的当地限制、习俗、街区安全、营业时间和预约变化;规划考虑距离、公共交通、" + "开放时间、预约和疲劳程度的每日路线;平衡住宿、交通、饮食和门票费用,提供不同价位的替代方案," + "并主动避开游客陷阱。最终交付必须是排版清楚的六部分旅行简报,每天的安排不超过180个中文字。" + "这份简报会直接发给同行人,最终交付必须由独立的内容编排与视觉表达职责统一格式和信息层级," + "不能由路线规划职责顺手兼任。" + "我有膝盖旧伤,每日步行不得超过12000步,连续步行不得超过30分钟;不要安排红眼航班,任何换乘" + "不得少于90分钟。这些是硬性限制,不能只当作建议。缺少目的地、日期、总预算、同行人或节奏偏好" + "中的任何一项时,必须先说明缺少什么,不得开始规划。未经我明确批准,不得预订或付款;任何预订" + "或付款操作必须携带 approved=true,且单笔人民币金额不得超过2800元,否则必须拒绝。行程批准后" + "可以监控预订截止时间和余位变化,但非紧急提醒只能在我的当地时间18:00到21:00发送。如果查询" + "失败或信息不足,必须返回已经确认的事实、仍缺少的信息和下一步,不得编造。现在只创建并保存" + "这个助手,不要规划任何具体旅行,也不要执行工作流。" +) + +pytestmark = [ + pytest.mark.real_llm, + pytest.mark.slow, + pytest.mark.skipif(not _CODEX_AUTH.is_file(), reason="no Codex OAuth credential"), +] + + +def _provider(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> OpenAICodexProvider: + raw = json.loads(_CODEX_AUTH.read_text(encoding="utf-8")) + tokens = raw.get("tokens") or {} + if not tokens.get("access_token") or not tokens.get("refresh_token"): + pytest.skip("Codex OAuth file has no usable tokens") + token_dir = tmp_path / "codex-oauth" + token_dir.mkdir(mode=0o700) + auth = { + "access_token": tokens["access_token"], + "refresh_token": tokens["refresh_token"], + "id_token": tokens.get("id_token"), + "account_id": tokens.get("account_id"), + } + path = token_dir / "auth.json" + path.write_text(json.dumps(auth), encoding="utf-8") + path.chmod(0o600) + monkeypatch.setenv("CHATGPT_TOKEN_DIR", str(token_dir)) + monkeypatch.setenv("CHATGPT_AUTH_FILE", "auth.json") + return OpenAICodexProvider(default_model=MODEL) + + +@pytest.mark.asyncio +async def test_live_generators_keep_task_and_persona_contracts_separate( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + provider = _provider(tmp_path, monkeypatch) + roster = ["Raven"] + profiles = {"Raven": {"description": "general worker that can research, reason, and use local tools"}} + + task = await TaskPlaybookGenerator(provider, MODEL).resolve( + "Create a reusable primary-source due-diligence task setup for evaluating a company.", + roster, + ["spawn", "run_subagent_dag"], + profiles, + ) + assert task.disposition == "runtime_and_artifact" + assert task.spec is not None and task.spec.delegate + assert task.capture_workflow + assert all(entry.playbook is None or not entry.playbook.memory.system_prompt for entry in task.spec.delegate) + + persona = await PersonaPlaybookGenerator(provider, MODEL).resolve( + "Create and save a reusable digital persona named claim-auditor. " + "It skeptically audits factual claims, requires primary sources, and flags uncertainty. " + "Do not run research now and do not create a workflow.", + roster, + ["spawn", "run_subagent_dag"], + profiles, + ) + assert persona.disposition == "artifact" + assert persona.spec is not None and persona.spec.delegate + assert persona.artifact_name + assert not persona.capture_workflow + + +@pytest.mark.asyncio +async def test_live_compiler_parameterizes_an_accepted_harness_dag( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + provider = _provider(tmp_path, monkeypatch) + harness = AgentPlaybookSpec( + name="launch-evidence-brief", + description="Evidence-first launch research", + delegate=[ + DelegateEntry( + **{ + "as": "researcher", + "name": "Raven", + "brief": "Use primary sources and identify uncertainty", + } + ) + ], + ) + dag = SubAgentDagSpec( + taskSummary="Build launch evidence brief", + confirm=False, + nodes=[ + DagNodeSpec( + id="research", + subagent="researcher", + nodeSummary="Research Acme launch", + promptTemplate="Research Acme's launch claims using primary sources", + ), + DagNodeSpec( + id="brief", + subagent="researcher", + nodeSummary="Write evidence brief", + promptTemplate="Turn {{ research.output }} into a concise evidence brief", + dependsOn=["research"], + ), + ], + ) + + compiled = await WorkflowCompiler(provider, MODEL).compile( + query="Create a reusable evidence brief process for Acme launch claims", + dag=dag, + run_id="live-e2e-run", + harness=harness, + name_hint="launch-evidence-brief", + ) + + assert compiled.harness is not None and compiled.workflow is not None + assert [node.id for node in compiled.workflow.nodes] == ["research", "brief"] + assert compiled.workflow.nodes[1].depends_on == ["research"] + assert compiled.metadata.source_run_id == "live-e2e-run" + + +@pytest.mark.asyncio +async def test_live_whole_turn_saves_a_persona_without_a_workflow( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + provider = _provider(tmp_path, monkeypatch) + playbook_root = tmp_path / "playbooks" + loop = AgentLoop( + provider=provider, + workspace=tmp_path, + model=MODEL, + policy=TurnPolicy(max_iterations=3), + tools=ToolWiring(restrict_to_workspace=True), + engine=EngineWiring( + runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), + playbook_config=PlaybookConfig(enabled=True, dir=str(playbook_root), agentHarness="generate"), + ), + ) + + async def emit(*args, **kwargs) -> None: + return None + + await loop.run_turn( + TurnRequest( + origin=Origin.USER, + source=Source(channel="test", chat_id="live-persona", sender_id="user", chat_type=ChatType.DM), + text=( + "Create and save a reusable digital persona called source-skeptic. " + "It challenges unsupported claims, demands primary evidence, and states uncertainty. " + "Do not execute a task and do not create a workflow." + ), + conversation="test:live-persona", + playbook_mode="persona", + ), + emit, + lambda: [], + stream=False, + ) + + names = [name for name in PlaybookStore(playbook_root).list_ids()] + assert len(names) == 1 + saved = PlaybookStore(playbook_root).load(names[0]) + assert isinstance(saved, UnifiedPlaybookSpec) + assert saved.harness is not None and saved.workflow is None + assert saved.harness.delegate + assert list((playbook_root / ".runs").glob("*.json")) + assert not list(tmp_path.glob("skills/**/SKILL.md")), "Persona mode must not duplicate the Harness as a skill" + + +@pytest.mark.asyncio +async def test_live_whole_turn_infers_a_travel_assistant_harness_from_user_needs( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + provider = _provider(tmp_path, monkeypatch) + playbook_root = tmp_path / "playbooks" + loop = AgentLoop( + provider=provider, + workspace=tmp_path, + model=MODEL, + policy=TurnPolicy(max_iterations=3), + tools=ToolWiring(plugin_tools=[LoadPlaybookTool()], restrict_to_workspace=True), + engine=EngineWiring( + runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), + playbook_config=PlaybookConfig(enabled=True, dir=str(playbook_root), agentHarness="generate"), + ), + subagents=SubagentWiring( + agents=[ + BuiltinAgentConfig( + name="Raven-Research", + description="Researches current external facts and verifies primary sources.", + owns="live web research, source verification, restrictions, safety, prices, and opening hours", + ), + BuiltinAgentConfig( + name="Raven-Design", + description="Turns approved content into polished, structured visual deliverables.", + owns="visual hierarchy, document structure, concise layout, and presentation quality", + ), + BuiltinAgentConfig( + name="Raven-OnCall", + description="Monitors changing conditions and sends time-sensitive updates.", + owns="watched work, availability changes, deadlines, and scheduled reminders", + ), + ] + ), + ) + + async def emit(*args, **kwargs) -> None: + return None + + reply: dict[str, object] = {} + await loop.run_turn( + TurnRequest( + origin=Origin.USER, + source=Source(channel="test", chat_id="live-travel-assistant", sender_id="user", chat_type=ChatType.DM), + text=TRAVEL_PERSONA_PROMPT, + conversation="test:live-travel-assistant", + playbook_mode="persona", + ), + emit, + lambda: [], + stream=False, + text_sink=reply, + ) + + names = PlaybookStore(playbook_root).list_ids() + assert len(names) == 1 + saved = PlaybookStore(playbook_root).load(names[0]) + assert isinstance(saved, UnifiedPlaybookSpec) + assert saved.harness is not None and saved.workflow is None + workers = saved.harness.delegate + assert workers + assert {"Raven-Research", "Raven-Design", "Raven-OnCall"} <= {worker.name for worker in workers} + assert any(worker.label != worker.name for worker in workers) + assert any(worker.playbook and worker.playbook.memory.functions.get("intake") for worker in workers) + assert any( + worker.playbook and worker.playbook.action.checks and worker.playbook.action.checks.code for worker in workers + ) + assert all("save the assistant" not in worker.brief.lower() for worker in workers) + assert all("do not plan a specific trip" not in worker.brief.lower() for worker in workers) + assert all("创建并保存" not in worker.brief for worker in workers) + assert all("不要规划任何具体旅行" not in worker.brief for worker in workers) + assert list((playbook_root / ".runs").glob("*.json")) + assert not list(tmp_path.glob("skills/**/SKILL.md")), "Persona mode must not duplicate the Harness as a skill" + + print( + json.dumps( + { + "reply": reply.get("text"), + "playbook": saved.name, + "artifactKind": "harness", + "harness": saved.harness.model_dump(by_alias=True, exclude_none=True), + "workflow": None, + }, + ensure_ascii=False, + indent=2, + ) + ) + + +@pytest.mark.asyncio +async def test_live_whole_turn_executes_and_saves_a_composite_playbook( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + provider = _provider(tmp_path, monkeypatch) + playbook_root = tmp_path / "playbooks" + loop = AgentLoop( + provider=provider, + workspace=tmp_path, + model=MODEL, + policy=TurnPolicy(max_iterations=5), + tools=ToolWiring(restrict_to_workspace=True), + engine=EngineWiring( + runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), + playbook_config=PlaybookConfig(enabled=True, dir=str(playbook_root), agentHarness="generate"), + ), + ) + + async def emit(*args, **kwargs) -> None: + return None + + await loop.run_turn( + TurnRequest( + origin=Origin.USER, + source=Source(channel="test", chat_id="live-composite", sender_id="user", chat_type=ChatType.DM), + text=( + "Create, execute now, and save a reusable Playbook named launch-signal-brief. " + "Use one complete run_subagent_dag graph for the reusable process. " + "Step 1 extracts exactly three launch signals from this passage: " + "'Northstar shipped offline mode, reduced cold-start latency by 35 percent, " + "and opened an EU support hub.' " + "Step 2 turns those signals into a concise executive brief. " + "Generate task-specific workers when useful, and save both their Harness and " + "the successful Workflow for later reuse." + ), + conversation="test:live-composite", + ), + emit, + lambda: [], + stream=False, + ) + + names = PlaybookStore(playbook_root).list_ids() + assert len(names) == 1 + assert {path.name for path in (playbook_root / names[0]).iterdir()} == {"playbook.md"} + saved = PlaybookStore(playbook_root).load(names[0]) + assert isinstance(saved, UnifiedPlaybookSpec) + assert saved.harness is not None and saved.workflow is not None + assert len(saved.workflow.nodes) >= 2 + assert {node.subagent for node in saved.workflow.nodes} <= {entry.label for entry in saved.harness.delegate} + records = list((playbook_root / ".runs").glob("*.json")) + assert len(records) == 1 + record = json.loads(records[0].read_text(encoding="utf-8")) + assert record["status"] == "completed" + assert len(record["dags"]) == 1 + assert record["savedPlaybook"] == names[0] diff --git a/tests/test_agent_playbook_delegate.py b/tests/test_agent_playbook_delegate.py index 43ea9b401..c7280dce5 100644 --- a/tests/test_agent_playbook_delegate.py +++ b/tests/test_agent_playbook_delegate.py @@ -25,7 +25,15 @@ from raven.agent.subagent.prompt_errors import DagValidationError from raven.config.schema import PlaybookConfig from raven.playbook import NodeSpec, PlaybookSpec, Triggers -from raven.playbook.agent_generator import WorkerTableGenerator, build_table, emit_tool, render_charter +from raven.playbook.agent_generator import ( + PERSONA_SYSTEM_PROMPT, + TASK_SYSTEM_PROMPT, + WorkerTableGenerator, + build_table, + persona_tool, + render_charter, + task_tool, +) from raven.playbook.agent_spec import AgentPlaybookSpec from raven.providers.base import LLMProvider, LLMResponse @@ -451,13 +459,35 @@ def test_the_brief_stands_in_when_no_prompt_was_written() -> None: # --------------------------------------------------------------------------- # +def test_task_and_persona_generation_have_distinct_instructions() -> None: + task = " ".join(TASK_SYSTEM_PROMPT.split()) + persona = " ".join(PERSONA_SYSTEM_PROMPT.split()) + + assert "one task instance, not a persona specification" in task + assert "Do not invent a DAG here" in task + assert "digital-person Harness" in persona + assert "Do not design a Workflow" in persona + assert "participant functions" not in task.lower() + assert "participant functions" in persona.lower() + + def test_the_roster_and_the_tools_are_enums_not_prose() -> None: """A name the host cannot resolve is refused at the boundary rather than diagnosed after, which is what keeps it out of the repair budget.""" - schema = emit_tool(["Raven-Research"], ["web_search"])[0]["function"]["parameters"] + schema = task_tool(["Raven-Research"], ["web_search"])[0]["function"]["parameters"] worker = schema["properties"]["workers"]["items"]["properties"] - assert worker["name"]["enum"] == ["Raven-Research"] + assert worker["agent"]["enum"] == ["Raven-Research"] assert worker["tools"]["items"]["enum"] == ["web_search"] + assert set(worker) <= {"as", "agent", "prompt", "tools"} + + +def test_persona_tool_exposes_rich_harness_fields_and_one_function_surface() -> None: + schema = persona_tool(["Raven-Research"], ["web_search"])[0]["function"]["parameters"] + worker = schema["properties"]["workers"]["items"]["properties"] + + assert {"agent", "brief", "systemPrompt", "stopWhen", "functions"} <= set(worker) + assert set(worker["functions"]["properties"]) == {"intake", "advise", "judge", "salvage"} + assert "code" not in worker def test_a_worker_off_the_roster_is_dropped_not_repaired() -> None: @@ -631,10 +661,8 @@ async def test_the_setup_call_runs_on_the_binding_it_was_handed(workspace) -> No def test_the_roster_reaches_the_generator_with_what_each_agent_is_for(workspace) -> None: """An enum of names is only selectable when the names say what they are. The shipped roster reads that way; a deployment's own does not.""" - from raven.playbook.agent_generator import emit_tool - - schema = emit_tool(["alpha", "beta"], ["grep"], {"alpha": "owns legal research", "beta": "owns code review"}) - described = schema[0]["function"]["parameters"]["properties"]["workers"]["items"]["properties"]["name"] + schema = persona_tool(["alpha", "beta"], ["grep"], {"alpha": "owns legal research", "beta": "owns code review"}) + described = schema[0]["function"]["parameters"]["properties"]["workers"]["items"]["properties"]["agent"] assert described["enum"] == ["alpha", "beta"] assert "owns legal research" in described["description"] @@ -642,11 +670,9 @@ def test_the_roster_reaches_the_generator_with_what_each_agent_is_for(workspace) def test_a_roster_that_says_nothing_still_renders(workspace) -> None: - from raven.playbook.agent_generator import emit_tool - - described = emit_tool(["alpha"], ["grep"])[0]["function"]["parameters"]["properties"]["workers"]["items"][ + described = persona_tool(["alpha"], ["grep"])[0]["function"]["parameters"]["properties"]["workers"]["items"][ "properties" - ]["name"] + ]["agent"] assert described["description"] == "Which sub-agent this worker is." @@ -671,14 +697,14 @@ def test_two_agents_with_blank_descriptions_are_still_told_apart(workspace) -> N is about, and a generator shown only names can drop the one that can. ``spawn`` gates on these same capabilities, which is why they are shown. """ - from raven.playbook.agent_generator import emit_tool, roster_note + from raven.playbook.agent_generator import roster_note metas = [_Meta("alpha"), _Meta("beta", stateful=True, reads_local_files=True, live_progress=True)] notes = {m.name: roster_note(m) for m in metas} - described = emit_tool(["alpha", "beta"], ["grep"], notes)[0]["function"]["parameters"]["properties"]["workers"][ + described = persona_tool(["alpha", "beta"], ["grep"], notes)[0]["function"]["parameters"]["properties"]["workers"][ "items" - ]["properties"]["name"]["description"] + ]["properties"]["agent"]["description"] assert "beta: reads local files" in described assert "resumable across dispatches" in described diff --git a/tests/test_cli_playbook_commands.py b/tests/test_cli_playbook_commands.py index 5316bf981..a56b23e96 100644 --- a/tests/test_cli_playbook_commands.py +++ b/tests/test_cli_playbook_commands.py @@ -125,7 +125,7 @@ class _FakeGenerator: def __init__(self, *args, **kwargs): pass - async def generate(self, user_input: str, skills=None) -> GeneratedPlaybook: + async def generate(self, user_input: str, skills=None, *, dag_only=False) -> GeneratedPlaybook: spec = PlaybookSpec( name="placeholder", description="generated from: " + user_input.splitlines()[0][:40], @@ -161,6 +161,7 @@ def test_create_lands_in_the_user_layer_usable(library, monkeypatch): assert "name: weekly-scan" in text assert "Assumption: weekly cadence" in text assert "Usable now" in r.stdout + assert "schemaVersion: 2" in text assert "disable weekly-scan" in r.stdout # The bracketed capability name is the diagnostic: Rich must not eat it. assert "mcp[fs]" in r.output diff --git a/tests/test_harness_generation_capabilities.py b/tests/test_harness_generation_capabilities.py new file mode 100644 index 000000000..0ff5e5130 --- /dev/null +++ b/tests/test_harness_generation_capabilities.py @@ -0,0 +1,445 @@ +"""The checked-in harness capability document controls every generation boundary.""" + +from __future__ import annotations + +import json + +import pytest +from pydantic import ValidationError + +from raven.agent import harness_capabilities +from raven.agent.subagent.charter import parse +from raven.playbook.agent_generator import ( + _choice_profile, + _persona_spec_from_args, + _profile_payload, + _task_spec_from_args, + _tool_inventory, + emit_tool, + persona_tool, + task_tool, +) +from raven.playbook.agent_spec import AgentPlaybookSpec + + +def _disabled_document(tmp_path, monkeypatch): + document = json.loads(harness_capabilities._PATH.read_text(encoding="utf-8")) + root = document["harnessGeneration"] + for module, name in ( + ("memory", "systemPrompt"), + ("memory", "stopWhen"), + ("capability", "tools"), + ("action", "checks"), + ): + root[module]["parameters"]["items"][name]["enabled"] = False + for module, name in (("memory", "intake"), ("planning", "advise"), ("action", "judge"), ("action", "salvage")): + root[module]["functions"]["participant"]["items"][name]["enabled"] = False + path = tmp_path / "harness_generation.json" + path.write_text(json.dumps(document), encoding="utf-8") + monkeypatch.setattr(harness_capabilities, "_PATH", path) + + +def test_checked_in_document_exposes_only_enabled_harness_fields() -> None: + properties = persona_tool(["worker"], ["read_file"])[0]["function"]["parameters"]["properties"]["workers"]["items"][ + "properties" + ] + + assert {"systemPrompt", "stopWhen", "tools", "checks", "functions"} <= properties.keys() + assert "code" not in properties + assert set(properties["functions"]["properties"]) == {"intake", "advise", "judge", "salvage"} + assert "reasoningEffort" not in properties + + +def test_disabling_fields_removes_them_from_generation(tmp_path, monkeypatch) -> None: + _disabled_document(tmp_path, monkeypatch) + + properties = persona_tool(["worker"], ["read_file"])[0]["function"]["parameters"]["properties"]["workers"]["items"][ + "properties" + ] + + assert not ({"systemPrompt", "stopWhen", "tools", "checks", "functions"} & properties.keys()) + assert {"as", "agent", "brief", "timeoutSeconds"} <= properties.keys() + + +def test_task_schema_and_compatibility_alias_respect_disabled_tools(tmp_path, monkeypatch) -> None: + _disabled_document(tmp_path, monkeypatch) + + schema = task_tool(["worker"], ["read_file"]) + worker = schema[0]["function"]["parameters"]["properties"]["workers"]["items"] + + assert "tools" not in worker["properties"] + assert emit_tool(["worker"], ["read_file"], {"worker": {}}, {"old": "ignored"}) == schema + + +def test_generator_profiles_and_tool_inventory_normalize_safe_inputs() -> None: + class Choice: + id = "max" + label = "Maximum" + description = "Use deeper reasoning" + + assert _choice_profile({"value": "fast", "name": "Fast"}) == {"id": "fast", "name": "Fast"} + assert _choice_profile(Choice()) == { + "id": "max", + "name": "Maximum", + "description": "Use deeper reasoning", + } + assert _profile_payload(" owns current web research ") == {"description": "owns current web research"} + assert _tool_inventory( + [ + "write_file", + {"function": {"name": "read_file", "description": "Read a file"}}, + {"function": []}, + object(), + ] + ) == [ + {"name": "read_file", "description": "Read a file"}, + {"name": "write_file", "description": ""}, + ] + + +@pytest.mark.parametrize( + ("worker", "message"), + [ + ("worker", "worker must be an object"), + ({"agent": "worker", "prompt": "do it", "brief": "forbidden"}, "field.*not allowed"), + ({"agent": "missing", "prompt": "do it"}, "unknown agent"), + ({"agent": "worker", "prompt": ""}, "prompt must be non-empty"), + ({"agent": "worker", "prompt": "do it", "tools": "read_file"}, "tools must be an array"), + ({"agent": "worker", "prompt": "do it", "tools": ["missing"]}, "unknown tool"), + ], +) +def test_task_generator_rejects_unusable_workers(worker, message) -> None: + with pytest.raises(ValueError, match=message): + _task_spec_from_args( + {"workers": [worker]}, + {"worker"}, + {"read_file"}, + ) + + +def test_task_generator_requires_workers_and_rejects_disabled_tools(tmp_path, monkeypatch) -> None: + with pytest.raises(ValueError, match="workers must be a non-empty list"): + _task_spec_from_args({"workers": []}, {"worker"}) + + _disabled_document(tmp_path, monkeypatch) + with pytest.raises(ValueError, match="disabled harness field.*tools"): + _task_spec_from_args( + {"workers": [{"agent": "worker", "prompt": "do it", "tools": ["read_file"]}]}, + {"worker"}, + {"read_file"}, + ) + + +def test_persona_generator_assembles_judge_rules_and_known_tools() -> None: + judge = "def judge(name, params, prior):\n return []" + spec, _ = _persona_spec_from_args( + { + "workers": [ + { + "as": "watcher", + "agent": "worker", + "brief": "Watch booking deadlines", + "tools": ["web_search"], + "checks": [{"tool": "web_search"}], + "functions": {"judge": judge}, + } + ] + }, + {"worker"}, + {"web_search"}, + ) + + playbook = spec.delegate[0].playbook + assert playbook is not None + assert playbook.capability.tools == ["web_search"] + assert playbook.action.checks.code == judge + assert [rule.tool for rule in playbook.action.checks.rules] == ["web_search"] + + +def test_persona_generator_requires_object_workers_and_known_tools() -> None: + with pytest.raises(ValueError, match="workers must be a non-empty list"): + _persona_spec_from_args({"workers": []}, {"worker"}) + with pytest.raises(ValueError, match="worker must be an object"): + _persona_spec_from_args({"workers": ["worker"]}, {"worker"}) + with pytest.raises(ValueError, match="unknown tool"): + _persona_spec_from_args( + { + "workers": [ + { + "as": "watcher", + "agent": "worker", + "brief": "Watch booking deadlines", + "tools": ["missing"], + } + ] + }, + {"worker"}, + {"web_search"}, + ) + + +def test_disabling_fields_rejects_generated_and_stored_values(tmp_path, monkeypatch) -> None: + _disabled_document(tmp_path, monkeypatch) + worker = { + "agent": "worker", + "systemPrompt": "special instructions", + "stopWhen": "done", + "tools": ["read_file"], + "checks": [{"tool": "read_file"}], + "functions": {"judge": "def judge(name, params, prior):\n return []"}, + } + + with pytest.raises(ValueError, match="disabled harness field"): + _persona_spec_from_args({"workers": [worker]}, {"worker"}) + with pytest.raises(ValidationError, match="disabled harness field"): + AgentPlaybookSpec.model_validate( + { + "delegate": [ + { + "name": "worker", + "playbook": { + "memory": {"systemPrompt": "special instructions"}, + "stopWhen": "done", + "capability": {"tools": ["read_file"]}, + "action": { + "checks": { + "rules": [{"tool": "read_file"}], + "code": "def judge(name, params, prior):\n return []", + } + }, + }, + } + ] + } + ) + + +def test_disabling_fields_drops_them_at_the_worker_boundary(tmp_path, monkeypatch) -> None: + _disabled_document(tmp_path, monkeypatch) + + charter = parse( + { + "brief": "the dispatch brief", + "instructionAddendum": "generated instructions", + "stopWhen": "done", + "tools": ["read_file"], + "checks": [{"tool": "read_file"}], + "code": "def judge(name, params, prior):\n return []", + } + ) + + assert charter is not None + assert charter.prompt == "the dispatch brief" + assert charter.instruction_addendum == "" + assert charter.task_brief == "the dispatch brief" + assert charter.stop_when == "" + assert charter.tools is None + assert charter.checks == () + assert charter.code == "" + + +def test_a_non_boolean_switch_is_refused(tmp_path, monkeypatch) -> None: + document = json.loads(harness_capabilities._PATH.read_text(encoding="utf-8")) + document["harnessGeneration"]["memory"]["parameters"]["items"]["systemPrompt"]["enabled"] = "false" + path = tmp_path / "harness_generation.json" + path.write_text(json.dumps(document), encoding="utf-8") + monkeypatch.setattr(harness_capabilities, "_PATH", path) + + with pytest.raises(RuntimeError, match="enabled must be a boolean"): + persona_tool(["worker"], ["read_file"]) + + +def test_capability_loader_refuses_broken_and_undeclared_catalog_entries(tmp_path, monkeypatch) -> None: + checked_in = json.loads(harness_capabilities._PATH.read_text(encoding="utf-8")) + + def point_at(name: str, document) -> None: + path = tmp_path / name + body = document if isinstance(document, str) else json.dumps(document) + path.write_text(body, encoding="utf-8") + monkeypatch.setattr(harness_capabilities, "_PATH", path) + + point_at("bad-json.json", "{") + with pytest.raises(RuntimeError, match="invalid harness generation capability file"): + harness_capabilities.parameter_enabled("memory", "systemPrompt") + + point_at("bad-root.json", {"harnessGeneration": []}) + with pytest.raises(RuntimeError, match="harnessGeneration must be an object"): + harness_capabilities.parameter_enabled("memory", "systemPrompt") + + missing_parameter = json.loads(json.dumps(checked_in)) + del missing_parameter["harnessGeneration"]["memory"]["parameters"]["items"]["systemPrompt"] + point_at("missing-parameter.json", missing_parameter) + with pytest.raises(RuntimeError, match="not declared: memory.parameters.systemPrompt"): + harness_capabilities.parameter_enabled("memory", "systemPrompt") + + missing_function = json.loads(json.dumps(checked_in)) + del missing_function["harnessGeneration"]["memory"]["functions"]["participant"]["items"]["intake"] + point_at("missing-function.json", missing_function) + with pytest.raises(RuntimeError, match="not declared: memory.functions.participant.intake"): + harness_capabilities.function_enabled("memory", "participant", "intake") + + unwired_parameter = json.loads(json.dumps(checked_in)) + unwired_parameter["harnessGeneration"]["memory"]["parameters"]["items"]["temperature"] = {"enabled": True} + point_at("unwired-parameter.json", unwired_parameter) + with pytest.raises(RuntimeError, match="enabled but not wired: memory.parameters.temperature"): + harness_capabilities.parameter_enabled("memory", "systemPrompt") + + +def test_enabling_an_unwired_catalog_item_fails_loudly(tmp_path, monkeypatch) -> None: + document = json.loads(harness_capabilities._PATH.read_text(encoding="utf-8")) + document["harnessGeneration"]["action"]["functions"]["participant"]["items"]["review"]["enabled"] = True + path = tmp_path / "harness_generation.json" + path.write_text(json.dumps(document), encoding="utf-8") + monkeypatch.setattr(harness_capabilities, "_PATH", path) + + with pytest.raises(RuntimeError, match="enabled but not wired: action.functions.participant.review"): + persona_tool(["worker"], ["read_file"]) + + +@pytest.mark.parametrize( + ("functions", "message"), + [ + ([], "functions must be an object"), + ({"archive": "def archive(step):\n return None"}, "unknown generated participant function: archive"), + ({"intake": ""}, "functions.intake must be non-empty Python source"), + ({"intake": "x" * 8001}, "functions.intake exceeds 8000 characters"), + ], +) +def test_generator_refuses_malformed_function_payloads(functions, message) -> None: + with pytest.raises(ValueError, match=message): + _persona_spec_from_args({"workers": [{"agent": "worker", "functions": functions}]}, {"worker"}) + + +@pytest.mark.parametrize( + ("extra", "message"), + [ + ({"code": "def judge(name, params, prior):\n return []"}, "Persona worker field.*code"), + ({"tools": "read_file"}, "tools must be an array"), + ({"checks": {}}, "checks must be an array"), + ({"timeoutSeconds": 0}, "timeoutSeconds must be a positive integer"), + ], +) +def test_persona_generator_refuses_fields_it_cannot_preserve(extra, message) -> None: + worker = {"as": "worker", "agent": "worker", "brief": "do it", **extra} + with pytest.raises(ValueError, match=message): + _persona_spec_from_args({"workers": [worker]}, {"worker"}) + + +@pytest.mark.parametrize( + ("playbook", "message"), + [ + ( + {"memory": {"functions": {"archive": "def archive(step):\n return None"}}}, + "unknown generated participant function: memory.archive", + ), + ({"memory": {"functions": {"intake": ""}}}, "generated participant function is empty: memory.intake"), + ({"action": {"checks": {"impl": "custom"}}}, "disabled harness field.*action.checksImpl"), + ], +) +def test_stored_playbook_refuses_unknown_empty_and_disabled_fields(playbook, message) -> None: + with pytest.raises(ValidationError, match=message): + AgentPlaybookSpec.model_validate({"delegate": [{"name": "worker", "playbook": playbook}]}) + + +@pytest.mark.asyncio +async def test_enabled_functions_generate_bind_and_execute_through_module_composers() -> None: + from raven.agent.harness.participants import compose_advice, compose_intake, compose_salvage + from raven.agent.subagent.charter import charter_participants, charter_scope + from raven.contracts.participant import StepView + from raven.playbook.agent_generator import build_payload + + sources = { + "intake": ("def intake(text, step):\n return {'text': text.upper(), 'note': step.get('phase')}"), + "advise": ("def advise(step):\n return 'iteration=' + str(step.get('iteration'))"), + "salvage": ("def salvage(step):\n return 'salvaged:' + step.get('question', '')"), + } + spec, briefs = _persona_spec_from_args( + {"workers": [{"as": "worker", "agent": "worker", "brief": "do it", "functions": sources}]}, + {"worker"}, + ) + playbook = spec.delegate[0].playbook + assert playbook is not None + payload = build_payload(briefs["worker"], playbook) + assert payload is not None and payload["functions"] == sources + charter = parse(payload) + assert charter is not None and dict(charter.functions) == sources + + step = StepView( + session_key="s", + iteration=3, + response=None, + transcript=({"role": "user", "content": "original"},), + history=(), + turn_base=0, + question="recover me", + rollbacks=0, + mode=None, + mode_overlay=None, + phase="user_inbound", + ) + with charter_scope(charter): + participants = charter_participants() + assert len(participants) == 1 + intake = await compose_intake("hello", step, participants) + assert intake is not None and intake.text == "HELLO" and intake.note == "user_inbound" + assert await compose_advice(step, participants) == "iteration=3" + assert await compose_salvage(step, participants) == "salvaged:recover me" + + +def test_generated_function_signature_is_validated_before_payload_building() -> None: + with pytest.raises(ValueError, match="must have signature intake\\(text, step\\)"): + _persona_spec_from_args( + {"workers": [{"agent": "worker", "functions": {"intake": "def intake(step):\n return None"}}]}, + {"worker"}, + ) + + +def test_disabled_generated_function_is_dropped_at_worker_boundary(tmp_path, monkeypatch) -> None: + _disabled_document(tmp_path, monkeypatch) + charter = parse( + { + "brief": "still present", + "functions": { + "intake": "def intake(text, step):\n return {'text': text}", + "advise": "def advise(step):\n return 'x'", + "salvage": "def salvage(step):\n return 'x'", + }, + } + ) + assert charter is not None + assert charter.functions == () + + +def test_disabling_one_function_removes_rejects_and_drops_only_that_function(tmp_path, monkeypatch) -> None: + document = json.loads(harness_capabilities._PATH.read_text(encoding="utf-8")) + document["harnessGeneration"]["planning"]["functions"]["participant"]["items"]["advise"]["enabled"] = False + path = tmp_path / "harness_generation.json" + path.write_text(json.dumps(document), encoding="utf-8") + monkeypatch.setattr(harness_capabilities, "_PATH", path) + + properties = persona_tool(["worker"], ["read_file"])[0]["function"]["parameters"]["properties"]["workers"]["items"][ + "properties" + ] + assert set(properties["functions"]["properties"]) == {"intake", "judge", "salvage"} + with pytest.raises(ValueError, match="disabled harness field.*functions.advise"): + _persona_spec_from_args( + {"workers": [{"agent": "worker", "functions": {"advise": "def advise(step):\n return 'x'"}}]}, + {"worker"}, + ) + + with pytest.raises(ValidationError, match="disabled harness field.*planning.functions.advise"): + AgentPlaybookSpec.model_validate( + {"delegate": [{"name": "worker", "playbook": {"planning": {"functions": {"advise": "x"}}}}]} + ) + + charter = parse( + { + "functions": { + "intake": "def intake(text, step):\n return {'text': text}", + "advise": "def advise(step):\n return 'x'", + "salvage": "def salvage(step):\n return 'x'", + } + } + ) + assert charter is not None + assert set(dict(charter.functions)) == {"intake", "salvage"} diff --git a/tests/test_playbook_tool.py b/tests/test_playbook_tool.py index c4b549e1b..017ad7b43 100644 --- a/tests/test_playbook_tool.py +++ b/tests/test_playbook_tool.py @@ -6,6 +6,8 @@ ``fills`` being able to complete a playbook but never to edit one. """ +import asyncio + import pytest from raven.agent.tools.load_playbook import LoadPlaybookTool @@ -133,6 +135,22 @@ def test_tool_schema_refreshes_after_adoption_in_the_same_turn(runtime): assert validate_params(loader.parameters, {"name": "monthly-feedback"}) == [] +async def test_preselected_playbook_is_isolated_between_concurrent_turns(tmp_path): + loader = LoadPlaybookTool(_runtime(tmp_path, [_spec(), _spec(name="monthly-feedback")])) + + async def render(selected: str) -> str: + loader.set_preselected(selected) + await asyncio.sleep(0) + return loader.description + + weekly, monthly = await asyncio.gather(render("weekly-feedback"), render("monthly-feedback")) + + assert "selected 'weekly-feedback'" in weekly + assert "selected 'monthly-feedback'" not in weekly + assert "selected 'monthly-feedback'" in monthly + assert "selected 'weekly-feedback'" not in monthly + + async def test_running_by_name_dispatches_the_same_graph(tmp_path): """The point of the tool: it joins the passive path at the executor, so a named run produces the dispatch a matched run would.""" @@ -173,7 +191,7 @@ def __init__(self, fail=False): self.fail = fail self.calls = [] - async def generate(self, workflow, skills=None): + async def generate(self, workflow, skills=None, *, dag_only=False): from raven.playbook import GeneratedPlaybook, PlaybookGenerationError self.calls.append((workflow, skills)) @@ -578,8 +596,8 @@ async def test_create_reports_a_name_written_while_generation_was_running(tmp_pa tool, store, adopted = _create_tool(tmp_path) generate = tool._generator.generate - async def _generate_after_another_writer(workflow, skills=None): - generated = await generate(workflow, skills) + async def _generate_after_another_writer(workflow, skills=None, *, dag_only=False): + generated = await generate(workflow, skills, dag_only=dag_only) store.save(_spec(name="weekly-scan", description="written by the winning request")) return generated diff --git a/tests/test_rpc_playbooks.py b/tests/test_rpc_playbooks.py index 1c32bc5ab..c47c88b02 100644 --- a/tests/test_rpc_playbooks.py +++ b/tests/test_rpc_playbooks.py @@ -593,8 +593,8 @@ async def test_validate_answers_its_findings_rather_than_raising( look like a broken call, and carries one string where this carries the list.""" library.save(_spec()) monkeypatch.setattr( - "raven.playbook.validate.validate_structure", - lambda spec, known_agents=None: ["merge: depends on a node that does not exist"], + "raven.playbook.runtime.validate_structure", + lambda spec, known_agents=None, **kwargs: ["merge: depends on a node that does not exist"], ) out = await mod.playbooks_validate({"name": "competitor-scan"}) @@ -612,7 +612,7 @@ async def test_validate_says_ok_with_an_empty_list_not_a_missing_one( """Paired with the case above so neither is satisfied by a handler that always answers the same shape.""" library.save(_spec()) - monkeypatch.setattr("raven.playbook.validate.validate_structure", lambda spec, known_agents=None: []) + monkeypatch.setattr("raven.playbook.runtime.validate_structure", lambda spec, known_agents=None, **kwargs: []) out = await mod.playbooks_validate({"name": "competitor-scan"}) @@ -1189,7 +1189,7 @@ def __init__(self, result=None, raises=None, hangs=False): self._raises = raises self._hangs = hangs - async def generate(self, workflow, skills=None): + async def generate(self, workflow, skills=None, *, dag_only=False): import asyncio self.calls.append((workflow, skills)) @@ -1370,3 +1370,56 @@ async def test_creating_with_no_runtime_refuses(library: PlaybookStore) -> None: def test_the_create_contract_is_mirrored_by_a_model_pair() -> None: assert "playbooks.create" in METHOD_MODELS + + +@pytest.mark.asyncio +async def test_frontend_rpc_exposes_unified_kind_workers_and_workflow(library: PlaybookStore) -> None: + """The old graph view stays populated while a new page can render Harness workers.""" + from raven.agent.subagent.dag_graph import DagNodeSpec + from raven.playbook.agent_spec import AgentPlaybookSpec, DelegateEntry + from raven.playbook.unified import PlaybookMatch, UnifiedPlaybookSpec, WorkflowSpec + + library.save( + UnifiedPlaybookSpec( + name="evidence-team", + description="Reusable evidence team and report process", + match=PlaybookMatch(summary="Build evidence report", keywords=["evidence report"]), + harness=AgentPlaybookSpec( + name="evidence-team", + description="Evidence workers", + delegate=[ + DelegateEntry( + **{ + "as": "researcher", + "name": "Raven", + "brief": "Use primary sources", + } + ) + ], + ), + workflow=WorkflowSpec( + summary="Build evidence report", + confirm=False, + nodes=[ + DagNodeSpec( + id="research", + subagent="researcher", + nodeSummary="Collect evidence", + promptTemplate="Research the subject", + ) + ], + ), + ) + ) + + row = (await mod.playbooks_list({}))["playbooks"][0] + assert row["schema_version"] == 2 + assert row["artifact_kind"] == "composite" + assert row["workers"] == [{"label": "researcher", "agent": "Raven"}] + assert row["nodes"] == [{"id": "research", "depends_on": []}] + METHOD_MODELS["playbooks.list"][1].model_validate({"playbooks": [row]}) + + detail = (await mod.playbooks_get({"name": "evidence-team"}))["playbook"] + assert detail["workers"] == [{"label": "researcher", "agent": "Raven", "brief": "Use primary sources"}] + assert detail["nodes"][0]["subagent"] == "researcher" + METHOD_MODELS["playbooks.get"][1].model_validate({"playbook": detail}) diff --git a/tests/test_rpc_turn_send.py b/tests/test_rpc_turn_send.py index fc20afe3e..e2ed01825 100644 --- a/tests/test_rpc_turn_send.py +++ b/tests/test_rpc_turn_send.py @@ -234,6 +234,27 @@ async def test_turn_send_accepts_optional_channel_chat_id_sender_id() -> None: assert (src.channel, src.chat_id, src.sender_id) == ("tui", "default", "user") +@pytest.mark.parametrize("mode", ["off", "task", "persona"]) +async def test_turn_send_passes_the_playbook_mode_to_the_turn(mode: str) -> None: + scheduler = FakeScheduler() + + await turn_send( + {"session_key": "tui:default", "content": "hi", "playbook_mode": mode}, + scheduler=scheduler, + turn_ids={}, + ) + + assert scheduler.submitted[0].playbook_mode == mode + + +async def test_turn_send_rejects_an_unknown_playbook_mode() -> None: + with pytest.raises(ValidationError): + await turn_send( + {"session_key": "tui:default", "content": "hi", "playbook_mode": "automatic"}, + scheduler=FakeScheduler(), + ) + + # --- End-to-end via Dispatcher --- diff --git a/tests/test_subagent_charter.py b/tests/test_subagent_charter.py index 43ed51459..427c62c9e 100644 --- a/tests/test_subagent_charter.py +++ b/tests/test_subagent_charter.py @@ -23,7 +23,9 @@ ) from raven.agent.subagent.charter_code import ( CharterCodeError, + compile_function, compile_judge, + run_function, run_judge, ) @@ -207,6 +209,30 @@ def test_a_judgement_about_one_call_compiles_and_runs() -> None: assert run_judge(fn, "grep", {}, []) == [] +def test_safe_numeric_signs_compile_and_run() -> None: + fn = compile_judge( + "def judge(name, params, prior):\n" + ' delta = -1 if params.get("late") else +1\n' + ' return ["late"] if delta < 0 else []' + ) + + assert run_judge(fn, "remind", {"late": True}, []) == ["late"] + assert run_judge(fn, "remind", {"late": False}, []) == [] + + +def test_safe_type_checks_compile_and_run() -> None: + fn = compile_judge( + "def judge(name, params, prior):\n" + ' amount = params.get("amount")\n' + " if not isinstance(amount, (int, float)):\n" + ' return ["amount must be numeric"]\n' + " return []" + ) + + assert run_judge(fn, "book", {"amount": "2800"}, []) == ["amount must be numeric"] + assert run_judge(fn, "book", {"amount": 2800.0}, []) == [] + + @pytest.mark.parametrize( ("label", "source"), [ @@ -236,6 +262,12 @@ def test_an_over_long_source_is_refused_before_parsing() -> None: compile_judge("def judge(n, p, q):\n return []\n" + "# padding\n" * 5000) +def test_generic_function_compilation_rejects_unknown_verbs_and_runtime_errors_are_silence() -> None: + with pytest.raises(CharterCodeError, match="unsupported charter function"): + compile_function("def archive(step):\n return None", "archive") + assert run_function(lambda value: 1 / 0, {"detached": True}) is None + + def test_a_judge_cannot_reach_a_name_it_was_not_given() -> None: """The namespace is an allow-list, not a denylist: a denylist is only as complete as the day it was written.""" @@ -427,6 +459,21 @@ def test_a_dispatch_brings_its_judgements_as_a_participant() -> None: ] +def test_a_refused_generated_function_is_cached_as_silence(monkeypatch) -> None: + from raven.agent.subagent import charter as charter_mod + + source = "def advise(step):\n import os\n return 'never'" + charter_mod._COMPILED.clear() + monkeypatch.setattr(charter_mod, "MAX_COMPILED", 0) + try: + with charter_scope(Charter(functions=(("advise", source),))): + assert charter_mod._generated_answer("advise", {}) is None + assert charter_mod._generated_answer("advise", {}) is None + assert charter_mod._COMPILED[("advise", source)] is False + finally: + charter_mod._COMPILED.clear() + + def test_a_plugins_own_rules_speak_before_the_dispatchs() -> None: """Participant order is product first: a veto needs one voice, and the wording the model reads should be the product's own where both would diff --git a/tests/test_subagent_manager.py b/tests/test_subagent_manager.py index a9b98b651..443bd2e7a 100644 --- a/tests/test_subagent_manager.py +++ b/tests/test_subagent_manager.py @@ -941,6 +941,26 @@ def _spy_init(self, workspace, *args, **kwargs): assert calls == [False] +def test_generated_participant_notes_preserve_each_supported_message_shape() -> None: + from raven.agent.subagent.backends.raven_loop import _append_participant_note + + absent: list[dict] = [] + _append_participant_note(absent, "note") + assert absent == [] + + text = [{"content": "base"}] + _append_participant_note(text, "note") + assert text[-1]["content"] == "base\n\nnote" + + blocks = [{"content": [{"type": "text", "text": "base"}]}] + _append_participant_note(blocks, "note") + assert blocks[-1]["content"][-1] == {"type": "text", "text": "note"} + + empty = [{"content": None}] + _append_participant_note(empty, "note") + assert empty[-1]["content"] == "note" + + def test_build_subagent_prompt_hides_orchestration_skills(tmp_path): """The catalog handed to a sub-agent must not advertise a skill whose procedure needs a tool this backend never registers — the DAG skill tells diff --git a/tests/test_unified_playbook.py b/tests/test_unified_playbook.py new file mode 100644 index 000000000..db96736f7 --- /dev/null +++ b/tests/test_unified_playbook.py @@ -0,0 +1,530 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import pytest +from pydantic import ValidationError + +from raven.agent.subagent.dag_adjudication import Final, Report +from raven.agent.subagent.dag_graph import DagNodeSpec, SubAgentDagSpec +from raven.agent.subagent.dag_runner import DagRunResult +from raven.agent.subagent.dag_tool import _successful_final +from raven.agent.subagent.delegate import current_delegate, delegate_scope +from raven.config.schema import PlaybookConfig +from raven.playbook.agent_generator import ( + PERSONA_TOOL, + TASK_TOOL, + PersonaPlaybookGenerator, + TaskPlaybookGenerator, + participant_function_syntax_guide, +) +from raven.playbook.agent_spec import AgentPlaybookSpec, DelegateEntry +from raven.playbook.runtime import PlaybookRuntime +from raven.playbook.store import PlaybookStore +from raven.playbook.types import ParamSpec +from raven.playbook.unified import InputSchema, PlaybookMatch, UnifiedPlaybookSpec, WorkflowSpec +from raven.playbook.workflow_compiler import EMIT_WORKFLOW, WorkflowCompiler +from raven.providers.base import LLMResponse, ToolCallRequest + + +class _Provider: + def __init__(self, *responses: LLMResponse) -> None: + self.responses = list(responses) + self.calls: list[dict] = [] + + async def chat_with_retry(self, **kwargs): + self.calls.append(kwargs) + return self.responses.pop(0) + + +def _call(name: str, arguments: dict) -> LLMResponse: + return LLMResponse( + content="", + tool_calls=[ToolCallRequest(id="call", name=name, arguments=arguments)], + finish_reason="tool_calls", + ) + + +def _harness() -> AgentPlaybookSpec: + return AgentPlaybookSpec( + name="research-team", + description="A focused research team", + delegate=[ + DelegateEntry( + **{ + "as": "analyst", + "name": "Raven", + "brief": "Collect evidence only", + "playbook": {"memory": {"systemPrompt": "Cite every claim."}}, + } + ) + ], + ) + + +def _workflow() -> WorkflowSpec: + return WorkflowSpec( + summary="Research and report", + confirm=False, + nodes=[ + DagNodeSpec( + id="research", + subagent="analyst", + node_summary="Collect evidence", + prompt_template="Research the topic", + ) + ], + ) + + +def test_playbook_generation_mode_defaults_and_camel_case_config() -> None: + assert PlaybookConfig().default_generation_mode == "task" + assert PlaybookConfig(defaultGenerationMode="persona").default_generation_mode == "persona" + with pytest.raises(ValidationError): + PlaybookConfig(defaultGenerationMode="automatic") + + +def test_persona_function_syntax_guide_matches_the_safety_boundary() -> None: + guide = participant_function_syntax_guide() + + assert "isinstance" in guide["allowedCalls"] + assert "append or any unlisted call or method" in guide["forbidden"] + assert "for and while statements are forbidden" in guide["allowedIteration"] + + +def test_unified_composite_round_trips_through_the_existing_store(tmp_path: Path) -> None: + store = PlaybookStore(tmp_path / "user", builtin_root=tmp_path / "builtin") + spec = UnifiedPlaybookSpec( + name="research-team", + description="Research a topic with a reusable evidence worker", + match=PlaybookMatch(summary="Research a topic", keywords=["research topic", "evidence scan"]), + harness=_harness(), + workflow=_workflow(), + ) + + path = store.save(spec) + loaded = store.load(spec.name) + + assert loaded == spec + assert "schemaVersion: 2" in path.read_text(encoding="utf-8") + assert "Workers: analyst" in path.read_text(encoding="utf-8") + + +def test_generated_judge_round_trips_without_serializing_an_implicit_checks_impl(tmp_path: Path) -> None: + store = PlaybookStore(tmp_path / "user", builtin_root=tmp_path / "builtin") + judge = "def judge(name, params, prior):\n return []" + harness = AgentPlaybookSpec( + name="safe-booking", + description="Approve travel bookings safely", + delegate=[ + DelegateEntry( + **{ + "as": "booking-guard", + "name": "Raven", + "brief": "Reject unapproved bookings", + "playbook": {"action": {"checks": {"code": judge}}}, + } + ) + ], + ) + spec = UnifiedPlaybookSpec( + name="safe-booking", + description="Approve travel bookings safely", + match=PlaybookMatch(summary="Approve travel bookings", keywords=["travel booking"]), + harness=harness, + ) + + body = store.save(spec).read_text(encoding="utf-8") + loaded = store.load(spec.name) + + assert "impl: default" not in body + assert isinstance(loaded, UnifiedPlaybookSpec) + assert loaded.harness is not None + checks = loaded.harness.delegate[0].playbook.action.checks + assert checks is not None and checks.code == judge + + +@pytest.mark.asyncio +async def test_task_generator_receives_only_task_selection_inputs() -> None: + provider = _Provider( + _call( + TASK_TOOL, + { + "artifactName": "due-diligence", + "description": "Research a company before acquisition", + "workers": [{"agent": "Raven", "prompt": "Collect primary evidence", "tools": ["web_search"]}], + }, + ) + ) + + result = await TaskPlaybookGenerator(provider, "stub").resolve( + "Run due diligence on Acme", + ["Raven"], + [{"type": "function", "function": {"name": "web_search", "description": "Search the web"}}], + {"Raven": {"description": "general worker", "readsLocalFiles": True}}, + {"due-diligence": "Checks a company before acquisition"}, + ) + + assert result.active + assert result.selected_playbook is None + assert result.capture_workflow + assert result.table and result.table.get("Raven").brief == "Collect primary evidence" + payload = json.loads(provider.calls[0]["messages"][1]["content"]) + assert payload["agents"][0]["readsLocalFiles"] is True + assert payload["availableTools"] == [{"name": "web_search", "description": "Search the web"}] + assert "playbookCandidates" not in payload + + +@pytest.mark.asyncio +async def test_resolver_emits_a_durable_harness_with_its_brief() -> None: + provider = _Provider( + _call( + PERSONA_TOOL, + { + "description": "A citation-first research persona", + "artifactName": "citation-researcher", + "workers": [ + { + "as": "researcher", + "agent": "Raven", + "brief": "Find primary sources", + "systemPrompt": "Cite primary sources only.", + } + ], + }, + ) + ) + + result = await PersonaPlaybookGenerator(provider, "stub").resolve( + "Create a citation-first research digital persona", + ["Raven"], + ["web_search"], + ) + + assert result.disposition == "artifact" + assert result.artifact_name == "citation-researcher" + assert not result.capture_workflow + assert result.spec is not None + assert result.spec.delegate[0].brief == "Find primary sources" + assert result.table and result.table.get("researcher").brief == "Find primary sources" + instructions = provider.calls[0]["messages"][0]["content"] + assert "Workers are operating components of the persona" in instructions + assert "Do not design a Workflow" in instructions + + +@pytest.mark.asyncio +async def test_harness_only_load_binds_workers_for_the_rest_of_the_turn(tmp_path: Path) -> None: + class _Executor: + dag_tool = None + + store = PlaybookStore(tmp_path / "user", builtin_root=tmp_path / "builtin") + store.save( + UnifiedPlaybookSpec( + name="research-team", + description="Reusable evidence workers", + match=PlaybookMatch(summary="Evidence research", keywords=["evidence research"]), + harness=_harness(), + ) + ) + runtime = PlaybookRuntime(store=store, executor=_Executor(), known_agents=lambda: ["Raven"]) + + with delegate_scope(None): + plan = await runtime.load("research-team") + assert plan is not None and plan.kind == "guidance" + table = current_delegate() + assert table is not None + assert table.get("analyst").brief == "Collect evidence only" + assert current_delegate() is None + + +@pytest.mark.asyncio +async def test_workflow_compiler_builds_a_composite_v2_artifact() -> None: + dag = SubAgentDagSpec(task_summary="Research", confirm=False, nodes=_workflow().nodes) + emitted = { + "name": "research-team", + "description": "Research a topic with cited evidence", + "match": {"summary": "Research a topic", "keywords": ["research topic", "cited evidence"]}, + "inputSchema": {"type": "object", "properties": {}, "required": [], "additionalProperties": False}, + "workflow": { + "summary": "Research", + "confirm": False, + "nodes": [node.model_dump(by_alias=True) for node in dag.nodes], + }, + } + provider = _Provider(_call(EMIT_WORKFLOW, emitted)) + + compiled = await WorkflowCompiler(provider, "stub").compile( + query="Research Acme", + dag=dag, + run_id="pb-test", + harness=_harness(), + ) + + assert compiled.schema_version == 2 + assert compiled.harness is not None and compiled.workflow is not None + assert compiled.workflow.nodes[0].subagent == "analyst" + assert compiled.metadata.source_run_id == "pb-test" + schema = provider.calls[0]["tools"][0]["function"]["parameters"] + assert schema["properties"]["workflow"]["properties"]["nodes"]["items"] + + +@pytest.mark.asyncio +async def test_workflow_compiler_accepts_reversible_prompt_parameterization() -> None: + source = _workflow().nodes[0].model_copy(update={"prompt_template": "Research Acme"}) + dag = SubAgentDagSpec(task_summary="Research", confirm=False, nodes=[source]) + emitted = { + "name": "research-team", + "description": "Research a topic with cited evidence", + "match": {"summary": "Research a topic", "keywords": ["research topic"]}, + "inputSchema": { + "type": "object", + "properties": { + "topic": { + "type": "string", + "required": False, + "default": "Acme", + "description": "Topic to research", + } + }, + "required": [], + "additionalProperties": False, + }, + "workflow": { + "summary": "Research", + "confirm": False, + "nodes": [ + { + **source.model_dump(by_alias=True), + "promptTemplate": "Research ${params.topic}", + } + ], + }, + } + provider = _Provider(_call(EMIT_WORKFLOW, emitted)) + + compiled = await WorkflowCompiler(provider, "stub").compile( + query="Research Acme", + dag=dag, + run_id="pb-parameterized", + harness=_harness(), + ) + + assert len(provider.calls) == 1 + assert compiled.workflow is not None + assert compiled.workflow.nodes[0].prompt_template == "Research ${params.topic}" + assert compiled.input_schema.properties["topic"].default == "Acme" + + +def test_workflow_capture_accepts_only_a_clean_completed_dag() -> None: + complete = DagRunResult( + run_id="run-ok", + dir="/tmp/run-ok", + summary={"total": 2, "completed": 2, "failed": 0, "cancelled": 0, "skipped": 0}, + ) + failed = DagRunResult( + run_id="run-failed", + dir="/tmp/run-failed", + summary={"total": 2, "completed": 1, "failed": 1, "cancelled": 0, "skipped": 0}, + ) + + assert _successful_final(Final(complete)) + assert not _successful_final(Final(failed)) + assert not _successful_final(Final(complete, stopped=True)) + assert not _successful_final(Report(node_id="research", text="needs a decision")) + + +def test_v2_file_is_plain_yaml_and_contains_no_run_values(tmp_path: Path) -> None: + store = PlaybookStore(tmp_path / "user", builtin_root=tmp_path / "builtin") + spec = UnifiedPlaybookSpec( + name="research-team", + description="Reusable evidence workers", + match=PlaybookMatch(summary="Evidence research", keywords=["evidence research"]), + harness=_harness(), + ) + body = store.save(spec).read_text(encoding="utf-8") + assert json.loads(json.dumps(spec.model_dump(by_alias=True)))["schemaVersion"] == 2 + assert "sourceRunId: null" not in body + + +@pytest.mark.asyncio +async def test_workflow_compiler_rejects_semantically_invalid_model_output() -> None: + dag = SubAgentDagSpec(task_summary="Research", confirm=False, nodes=_workflow().nodes) + invalid = { + "name": "research-team", + "description": "Research a topic with cited evidence", + "match": {"summary": "Research a topic", "keywords": ["research topic"]}, + "inputSchema": { + "type": "object", + "properties": { + "topic": { + "type": "string", + "required": True, + "description": "Topic to research", + } + }, + "required": ["topic"], + "additionalProperties": False, + }, + "workflow": { + "summary": "Research", + "confirm": False, + "nodes": [ + { + **dag.nodes[0].model_dump(by_alias=True), + "inputs": {"topic": "${params.topic}"}, + } + ], + }, + } + provider = _Provider(_call(EMIT_WORKFLOW, invalid), _call(EMIT_WORKFLOW, invalid)) + + compiled = await WorkflowCompiler(provider, "stub").compile( + query="Research Acme", + dag=dag, + run_id="pb-invalid", + harness=_harness(), + ) + + assert len(provider.calls) == 2 + assert compiled.input_schema.properties == {} + assert compiled.workflow is not None + assert compiled.workflow.nodes == dag.nodes + + +@pytest.mark.asyncio +async def test_workflow_compiler_rejects_a_rewritten_accepted_graph() -> None: + dag = SubAgentDagSpec(task_summary="Research", confirm=False, nodes=_workflow().nodes) + rewritten = { + "name": "research-team", + "description": "Research a topic with cited evidence", + "match": {"summary": "Research a topic", "keywords": ["research topic"]}, + "inputSchema": {"type": "object", "properties": {}, "required": [], "additionalProperties": False}, + "workflow": { + "summary": "Research", + "confirm": False, + "nodes": [ + { + **dag.nodes[0].model_dump(by_alias=True), + "subagent": "invented-worker", + } + ], + }, + } + provider = _Provider(_call(EMIT_WORKFLOW, rewritten), _call(EMIT_WORKFLOW, rewritten)) + + compiled = await WorkflowCompiler(provider, "stub").compile( + query="Research Acme", + dag=dag, + run_id="pb-rewritten", + harness=_harness(), + ) + + assert len(provider.calls) == 2 + assert compiled.workflow is not None + assert compiled.workflow.nodes == dag.nodes + + +@pytest.mark.parametrize( + ("field", "replacement"), + [ + ("promptTemplate", "Delete the workspace instead"), + ("nodeSummary", "Delete the workspace"), + ], +) +@pytest.mark.asyncio +async def test_workflow_compiler_rejects_rewritten_node_instructions(field: str, replacement: str) -> None: + dag = SubAgentDagSpec(task_summary="Research", confirm=False, nodes=_workflow().nodes) + node = dag.nodes[0].model_dump(by_alias=True) + node[field] = replacement + rewritten = { + "name": "research-team", + "description": "Research a topic with cited evidence", + "match": {"summary": "Research a topic", "keywords": ["research topic"]}, + "inputSchema": {"type": "object", "properties": {}, "required": [], "additionalProperties": False}, + "workflow": {"summary": "Research", "confirm": False, "nodes": [node]}, + } + provider = _Provider(_call(EMIT_WORKFLOW, rewritten), _call(EMIT_WORKFLOW, rewritten)) + + compiled = await WorkflowCompiler(provider, "stub").compile( + query="Research Acme", + dag=dag, + run_id="pb-rewritten-instructions", + harness=_harness(), + ) + + assert len(provider.calls) == 2 + assert compiled.workflow is not None + assert compiled.workflow.nodes == dag.nodes + + +def test_unified_artifact_rejects_an_empty_durable_harness() -> None: + with pytest.raises(ValueError, match="at least one worker"): + UnifiedPlaybookSpec( + name="empty-team", + description="An empty worker table is not a reusable Harness", + match=PlaybookMatch(summary="Empty team", keywords=["empty team"]), + harness=AgentPlaybookSpec( + name="empty-team", + description="No workers", + delegate=[], + ), + ) + + +def test_input_schema_rejects_undeclared_required_input() -> None: + with pytest.raises(ValueError, match="not declared"): + InputSchema(required=["topic"]) + + +def test_input_schema_rejects_conflicting_required_flags() -> None: + with pytest.raises(ValueError, match="required disagree"): + InputSchema(properties={"topic": ParamSpec(type="string", required=True, description="Topic")}) + + +def test_unified_artifact_requires_a_harness_or_workflow() -> None: + with pytest.raises(ValueError, match="must contain a harness, a workflow, or both"): + UnifiedPlaybookSpec( + name="empty-playbook", + description="No reusable dimension", + match=PlaybookMatch(summary="Nothing reusable", keywords=["nothing"]), + ) + + +@pytest.mark.asyncio +async def test_workflow_compiler_falls_back_when_the_provider_fails() -> None: + class _FailingProvider(_Provider): + async def chat_with_retry(self, **kwargs): + raise RuntimeError("provider unavailable") + + dag = SubAgentDagSpec(task_summary="Research", confirm=False, nodes=_workflow().nodes) + compiled = await WorkflowCompiler(_FailingProvider(), "stub").compile( + query="Research Acme", + dag=dag, + run_id="pb-provider-failure", + harness=_harness(), + ) + + assert compiled.workflow is not None + assert compiled.workflow.nodes == dag.nodes + assert compiled.metadata.source_run_id == "pb-provider-failure" + + +def test_composite_workflow_may_mix_harness_aliases_and_registered_agents() -> None: + workflow = _workflow() + mixed = workflow.model_copy( + update={ + "nodes": [ + *workflow.nodes, + DagNodeSpec(id="report", subagent="Raven", node_summary="Write report", prompt_template="Write report"), + ] + } + ) + spec = UnifiedPlaybookSpec( + name="mixed-team", + description="Use a specialist and the default roster together", + match=PlaybookMatch(summary="Research and report", keywords=["research report"]), + harness=_harness(), + workflow=mixed, + ) + assert [node.subagent for node in spec.workflow.nodes] == ["analyst", "Raven"] diff --git a/tests/test_unified_playbook_e2e.py b/tests/test_unified_playbook_e2e.py new file mode 100644 index 000000000..a7dd6e3c0 --- /dev/null +++ b/tests/test_unified_playbook_e2e.py @@ -0,0 +1,560 @@ +"""Whole-turn E2E for resolve -> Harness DAG -> compile -> save.""" + +from __future__ import annotations + +import asyncio +from typing import Any + +import pytest + +from raven.agent.loop import AgentLoop +from raven.agent.loop.bundles import EngineWiring, ToolWiring, TurnPolicy +from raven.agent.subagent.dag_graph import DagNodeSpec +from raven.agent.tools.load_playbook import LoadPlaybookTool +from raven.config.raven import CheckpointConfig, RuntimeConfig +from raven.config.schema import PlaybookConfig +from raven.playbook.agent_generator import PERSONA_TOOL, TASK_TOOL +from raven.playbook.agent_spec import AgentPlaybookSpec, DelegateEntry +from raven.playbook.store import PlaybookStore +from raven.playbook.unified import PlaybookMatch, UnifiedPlaybookSpec, WorkflowSpec +from raven.playbook.workflow_compiler import EMIT_WORKFLOW +from raven.providers.base import LLMProvider, LLMResponse, ToolCallRequest +from raven.spine.message import ChatType, Source +from raven.spine.turn import Origin, TurnRequest + + +def _names(tools: Any) -> set[str]: + return {(tool.get("function", tool) or {}).get("name", "") for tool in tools or []} + + +class _WholeTurnProvider(LLMProvider): + def __init__(self) -> None: + super().__init__(api_key="test") + self.main_calls = 0 + self.setup_calls = 0 + self.compiler_calls = 0 + self.worker_calls = 0 + + def get_default_model(self) -> str: + return "stub" + + async def chat(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self._answer(messages, tools) + + async def chat_with_retry(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self._answer(messages, tools) + + async def _answer(self, messages, tools) -> LLMResponse: + names = _names(tools) + if TASK_TOOL in names: + self.setup_calls += 1 + return LLMResponse( + content="", + tool_calls=[ + ToolCallRequest( + id="setup", + name=TASK_TOOL, + arguments={ + "description": "A reusable evidence workflow", + "artifactName": "evidence-brief", + "workers": [ + { + "as": "researcher", + "agent": "Raven", + "prompt": "Return WORKER_EVIDENCE_OK after collecting the requested evidence", + } + ], + }, + ) + ], + finish_reason="tool_calls", + ) + if EMIT_WORKFLOW in names: + self.compiler_calls += 1 + return LLMResponse( + content="", + tool_calls=[ + ToolCallRequest( + id="compile", + name=EMIT_WORKFLOW, + arguments={ + "name": "model-tried-to-rename-the-artifact", + "description": "Produce a concise evidence brief", + "match": { + "summary": "Produce an evidence brief", + "keywords": ["evidence brief", "source research"], + }, + "inputSchema": { + "type": "object", + "properties": {}, + "required": [], + "additionalProperties": False, + }, + "workflow": { + "summary": "Build evidence brief", + "confirm": False, + "nodes": [ + { + "id": "research", + "subagent": "researcher", + "nodeSummary": "Collect evidence", + "promptTemplate": "Return WORKER_EVIDENCE_OK", + "dependsOn": [], + } + ], + }, + }, + ) + ], + finish_reason="tool_calls", + ) + if "run_subagent_dag" in names: + self.main_calls += 1 + if self.main_calls == 1: + return LLMResponse( + content="", + tool_calls=[ + ToolCallRequest( + id="dag", + name="run_subagent_dag", + arguments={ + "task_summary": "Build evidence brief", + "background": True, + "nodes": [ + { + "id": "research", + "subagent": "researcher", + "node_summary": "Collect evidence", + "prompt_template": "Return WORKER_EVIDENCE_OK", + "depends_on": [], + } + ], + }, + ) + ], + finish_reason="tool_calls", + ) + return LLMResponse(content="UNIFIED_PLAYBOOK_SAVED_OK", finish_reason="stop") + self.worker_calls += 1 + return LLMResponse(content="WORKER_EVIDENCE_OK", finish_reason="stop") + + +@pytest.mark.asyncio +async def test_whole_turn_saves_a_composite_ready_playbook(tmp_path) -> None: + playbook_root = tmp_path / "playbooks" + provider = _WholeTurnProvider() + loop = AgentLoop( + provider=provider, + workspace=tmp_path, + model="stub", + policy=TurnPolicy(max_iterations=4), + tools=ToolWiring(restrict_to_workspace=True), + engine=EngineWiring( + runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), + playbook_config=PlaybookConfig(enabled=True, dir=str(playbook_root), agentHarness="generate"), + ), + ) + + async def emit(*args, **kwargs) -> None: + return None + + await loop.run_turn( + TurnRequest( + origin=Origin.USER, + source=Source(channel="test", chat_id="unified", sender_id="user", chat_type=ChatType.DM), + text="Research a topic and save this reusable evidence process", + conversation="test:unified", + ), + emit, + lambda: [], + stream=False, + ) + + loaded = PlaybookStore(playbook_root).load("evidence-brief") + assert isinstance(loaded, UnifiedPlaybookSpec) + assert loaded.harness is not None and loaded.workflow is not None + assert loaded.workflow.nodes[0].subagent == "researcher" + assert provider.setup_calls == provider.compiler_calls == 1 + assert provider.worker_calls >= 1 # worker plus the optional DAG verdict call + records = list((playbook_root / ".runs").glob("*.json")) + assert len(records) == 1 + assert '"status": "completed"' in records[0].read_text(encoding="utf-8") + + +async def _emit(*args, **kwargs) -> None: + return None + + +def _request(text: str, chat_id: str, *, playbook_mode: str | None = None) -> TurnRequest: + return TurnRequest( + origin=Origin.USER, + source=Source(channel="test", chat_id=chat_id, sender_id="user", chat_type=ChatType.DM), + text=text, + conversation=f"test:{chat_id}", + playbook_mode=playbook_mode, + ) + + +def _saved_composite() -> UnifiedPlaybookSpec: + harness = AgentPlaybookSpec( + name="evidence-brief", + description="A citation-first evidence worker", + delegate=[ + DelegateEntry( + **{ + "as": "researcher", + "name": "Raven", + "brief": "Collect primary evidence and return the marker", + "playbook": {"memory": {"systemPrompt": "Return WORKER_EVIDENCE_OK when complete."}}, + } + ) + ], + ) + return UnifiedPlaybookSpec( + name="evidence-brief", + description="Produce a concise evidence brief from primary sources", + match=PlaybookMatch( + summary="Produce an evidence brief", + keywords=["evidence brief", "primary source research"], + ), + harness=harness, + workflow=WorkflowSpec( + summary="Build evidence brief", + confirm=False, + nodes=[ + DagNodeSpec( + id="research", + subagent="researcher", + nodeSummary="Collect evidence", + promptTemplate="Return WORKER_EVIDENCE_OK", + ) + ], + ), + ) + + +class _ReuseProvider(LLMProvider): + def __init__(self) -> None: + super().__init__(api_key="test") + self.setup_calls = 0 + self.main_calls = 0 + self.worker_calls = 0 + + def get_default_model(self) -> str: + return "stub" + + async def chat(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self._answer(tools) + + async def chat_with_retry(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self._answer(tools) + + async def _answer(self, tools) -> LLMResponse: + names = _names(tools) + if TASK_TOOL in names or PERSONA_TOOL in names: + self.setup_calls += 1 + return LLMResponse( + content="", + tool_calls=[ + ToolCallRequest( + id="select", + name=TASK_TOOL, + arguments={ + "description": "The saved evidence process is an exact match", + "disposition": "none", + "selectedPlaybook": "evidence-brief", + "captureWorkflow": False, + "workers": [], + }, + ) + ], + finish_reason="tool_calls", + ) + if "load_playbook" in names: + self.main_calls += 1 + if self.main_calls == 1: + return LLMResponse( + content="", + tool_calls=[ + ToolCallRequest( + id="load", + name="load_playbook", + arguments={"name": "evidence-brief", "params": {}}, + ) + ], + finish_reason="tool_calls", + ) + return LLMResponse(content="SAVED_PLAYBOOK_REUSED_OK", finish_reason="stop") + self.worker_calls += 1 + return LLMResponse(content="WORKER_EVIDENCE_OK", finish_reason="stop") + + +@pytest.mark.asyncio +async def test_off_mode_skips_generation_but_keeps_explicit_playbook_loading(tmp_path) -> None: + playbook_root = tmp_path / "playbooks" + PlaybookStore(playbook_root).save(_saved_composite()) + provider = _ReuseProvider() + loop = AgentLoop( + provider=provider, + workspace=tmp_path, + model="stub", + policy=TurnPolicy(max_iterations=4), + tools=ToolWiring(plugin_tools=[LoadPlaybookTool()], restrict_to_workspace=True), + engine=EngineWiring( + runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), + playbook_config=PlaybookConfig(enabled=True, dir=str(playbook_root), agentHarness="generate"), + ), + ) + + await loop.run_turn( + _request("Run my primary-source evidence brief for the launch", "reuse", playbook_mode="off"), + _emit, + lambda: [], + stream=False, + ) + await asyncio.gather(*list(loop._playbooks.dag_tool._runs.values()), return_exceptions=True) + + assert provider.setup_calls == 0 + assert provider.main_calls >= 2 + assert provider.worker_calls >= 1 + assert not (playbook_root / ".runs").exists() + + +class _PersonaProvider(LLMProvider): + def __init__(self) -> None: + super().__init__(api_key="test") + self.setup_calls = 0 + self.main_calls = 0 + self.compiler_calls = 0 + + def get_default_model(self) -> str: + return "stub" + + async def chat(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self._answer(tools) + + async def chat_with_retry(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self._answer(tools) + + async def _answer(self, tools) -> LLMResponse: + names = _names(tools) + if PERSONA_TOOL in names: + self.setup_calls += 1 + return LLMResponse( + content="", + tool_calls=[ + ToolCallRequest( + id="persona", + name=PERSONA_TOOL, + arguments={ + "description": "A skeptical claim-checking digital persona", + "artifactName": "skeptical-fact-checker", + "workers": [ + { + "as": "fact-checker", + "agent": "Raven", + "brief": "Challenge unsupported claims and require primary evidence", + "systemPrompt": "Be skeptical, concise, and cite primary evidence.", + "stopWhen": "Every material claim is supported or flagged", + } + ], + }, + ) + ], + finish_reason="tool_calls", + ) + if EMIT_WORKFLOW in names: + self.compiler_calls += 1 + raise AssertionError("a Harness-only persona must not compile a Workflow") + self.main_calls += 1 + return LLMResponse(content="PERSONA_CREATED_WITHOUT_ODD_DAG_OK", finish_reason="stop") + + +@pytest.mark.asyncio +async def test_digital_persona_saves_harness_only_without_inventing_a_dag(tmp_path) -> None: + playbook_root = tmp_path / "playbooks" + provider = _PersonaProvider() + loop = AgentLoop( + provider=provider, + workspace=tmp_path, + model="stub", + policy=TurnPolicy(max_iterations=3), + tools=ToolWiring(restrict_to_workspace=True), + engine=EngineWiring( + runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), + playbook_config=PlaybookConfig(enabled=True, dir=str(playbook_root), agentHarness="generate"), + ), + ) + + await loop.run_turn( + _request( + "Create a skeptical fact-checking digital persona; do not run a process", + "persona", + playbook_mode="persona", + ), + _emit, + lambda: [], + stream=False, + ) + + saved = PlaybookStore(playbook_root).load("skeptical-fact-checker") + assert isinstance(saved, UnifiedPlaybookSpec) + assert saved.harness is not None + assert saved.workflow is None + assert saved.harness.delegate[0].brief.startswith("Challenge unsupported") + assert provider.setup_calls == provider.main_calls == 1 + assert provider.compiler_calls == 0 + + +@pytest.mark.asyncio +async def test_persona_name_collision_gets_a_deterministic_numeric_suffix(tmp_path) -> None: + playbook_root = tmp_path / "playbooks" + provider = _PersonaProvider() + loop = AgentLoop( + provider=provider, + workspace=tmp_path, + model="stub", + policy=TurnPolicy(max_iterations=3), + tools=ToolWiring(restrict_to_workspace=True), + engine=EngineWiring( + runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), + playbook_config=PlaybookConfig(enabled=True, dir=str(playbook_root), agentHarness="generate"), + ), + ) + + for chat_id in ("persona-one", "persona-two"): + await loop.run_turn( + _request("Create the same skeptical fact-checking persona", chat_id, playbook_mode="persona"), + _emit, + lambda: [], + stream=False, + ) + + assert PlaybookStore(playbook_root).list_ids() == ["skeptical-fact-checker", "skeptical-fact-checker-2"] + suffixed = PlaybookStore(playbook_root).load("skeptical-fact-checker-2") + assert isinstance(suffixed, UnifiedPlaybookSpec) + assert suffixed.harness is not None and suffixed.harness.name == "skeptical-fact-checker-2" + + +class _TaskWithoutDagProvider(LLMProvider): + def __init__(self) -> None: + super().__init__(api_key="test") + self.setup_calls = 0 + self.main_calls = 0 + self.compiler_calls = 0 + + def get_default_model(self) -> str: + return "stub" + + async def chat(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self._answer(tools) + + async def chat_with_retry(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self._answer(tools) + + async def _answer(self, tools) -> LLMResponse: + names = _names(tools) + if TASK_TOOL in names: + self.setup_calls += 1 + return LLMResponse( + content="", + tool_calls=[ + ToolCallRequest( + id="task-harness", + name=TASK_TOOL, + arguments={ + "artifactName": "concise-answer-task", + "description": "Answer a bounded question concisely", + "workers": [{"agent": "Raven", "prompt": "Answer concisely with cited facts"}], + }, + ) + ], + finish_reason="tool_calls", + ) + if EMIT_WORKFLOW in names: + self.compiler_calls += 1 + raise AssertionError("a Task without a successful DAG must not compile a Workflow") + self.main_calls += 1 + return LLMResponse(content="TASK_FINISHED_WITHOUT_DAG_OK", finish_reason="stop") + + +@pytest.mark.asyncio +async def test_task_without_a_dag_keeps_the_immediately_saved_harness(tmp_path) -> None: + playbook_root = tmp_path / "playbooks" + provider = _TaskWithoutDagProvider() + loop = AgentLoop( + provider=provider, + workspace=tmp_path, + model="stub", + policy=TurnPolicy(max_iterations=2), + tools=ToolWiring(restrict_to_workspace=True), + engine=EngineWiring( + runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), + playbook_config=PlaybookConfig(enabled=True, dir=str(playbook_root), agentHarness="generate"), + ), + ) + + await loop.run_turn( + _request("Answer this bounded question without delegating", "task-no-dag"), + _emit, + lambda: [], + stream=False, + ) + + saved = PlaybookStore(playbook_root).load("concise-answer-task") + assert isinstance(saved, UnifiedPlaybookSpec) + assert saved.harness is not None and saved.workflow is None + assert provider.setup_calls == provider.main_calls == 1 + assert provider.compiler_calls == 0 + + +class _OffProvider(LLMProvider): + def __init__(self) -> None: + super().__init__(api_key="test") + self.setup_calls = 0 + self.main_calls = 0 + + def get_default_model(self) -> str: + return "stub" + + async def chat(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self._answer(tools) + + async def chat_with_retry(self, messages, tools=None, model=None, **kwargs) -> LLMResponse: + return await self._answer(tools) + + async def _answer(self, tools) -> LLMResponse: + if {TASK_TOOL, PERSONA_TOOL} & _names(tools): + self.setup_calls += 1 + raise AssertionError("the disabled Playbook feature made a setup model call") + self.main_calls += 1 + return LLMResponse(content="OFF_PATH_USES_DEFAULT_ROSTER_OK", finish_reason="stop") + + +@pytest.mark.asyncio +async def test_master_switch_off_has_no_resolution_record_or_library_side_effect(tmp_path) -> None: + playbook_root = tmp_path / "playbooks" + provider = _OffProvider() + loop = AgentLoop( + provider=provider, + workspace=tmp_path, + model="stub", + policy=TurnPolicy(max_iterations=2), + tools=ToolWiring(restrict_to_workspace=True), + engine=EngineWiring( + runtime_config=RuntimeConfig(checkpoint=CheckpointConfig(policy="never")), + playbook_config=PlaybookConfig(enabled=False, dir=str(playbook_root), agentHarness="generate"), + ), + ) + + await loop.run_turn( + _request("Answer normally with Playbooks disabled", "off"), + _emit, + lambda: [], + stream=False, + ) + + assert loop._playbooks is None + assert provider.setup_calls == 0 + assert provider.main_calls == 1 + assert not playbook_root.exists() diff --git a/ui-tui/src/i18n/messages.generated.ts b/ui-tui/src/i18n/messages.generated.ts index e8c631cd1..972f52d52 100644 --- a/ui-tui/src/i18n/messages.generated.ts +++ b/ui-tui/src/i18n/messages.generated.ts @@ -1743,6 +1743,7 @@ export const UI_TEXT: Record> = { 'gui.pb.disabled': 'off', 'gui.pb.unreadable': 'Could not read this file: {why}', 'gui.pb.shape_live': 'composed per run', + 'gui.pb.shape_harness': 'Harness · {n}', 'gui.pb.more_steps': '+{n}', 'gui.pb.reading': 'Reading...', 'gui.pb.lane': '@{handle} one session', @@ -1759,11 +1760,15 @@ export const UI_TEXT: Record> = { 'gui.pb.zoom_out': 'Zoom out', 'gui.pb.zoom_fit': 'Fit the whole graph', 'gui.pb.tab_graph': 'Graph', + 'gui.pb.tab_harness': 'Harness', 'gui.pb.tab_assembly': 'Assembly guide', 'gui.pb.tab_contract': 'Contract', 'gui.pb.f_mode': 'mode', 'gui.pb.f_version': 'spec version', 'gui.pb.f_confirm': 'before dispatch', + 'gui.pb.f_artifact': 'artifact', + 'gui.pb.sec_harness': 'Harness', + 'gui.pb.harness_only': 'This artifact provides reusable workers and has no stored Workflow.', 'gui.pb.sec_keywords': 'Trigger keywords', 'gui.pb.sec_params': 'Runtime inputs', 'gui.pb.col_name': 'name', @@ -3475,6 +3480,7 @@ export const UI_TEXT: Record> = { 'gui.pb.disabled': '已停用', 'gui.pb.unreadable': '这个文件读不出来:{why}', 'gui.pb.shape_live': '流程每次现搭', + 'gui.pb.shape_harness': 'Harness · {n}', 'gui.pb.more_steps': '+{n}', 'gui.pb.reading': '正在读取…', 'gui.pb.lane': '@{handle} 同一会话', @@ -3491,11 +3497,15 @@ export const UI_TEXT: Record> = { 'gui.pb.zoom_out': '缩小', 'gui.pb.zoom_fit': '缩放到完整流程', 'gui.pb.tab_graph': '图', + 'gui.pb.tab_harness': 'Harness', 'gui.pb.tab_assembly': '编排指引', 'gui.pb.tab_contract': '契约', 'gui.pb.f_mode': '模式', 'gui.pb.f_version': '规范版本', 'gui.pb.f_confirm': '派发前确认', + 'gui.pb.f_artifact': '产物', + 'gui.pb.sec_harness': 'Harness', + 'gui.pb.harness_only': '这个产物提供可复用的 workers,不包含已保存的 Workflow。', 'gui.pb.sec_keywords': '触发关键词', 'gui.pb.sec_params': '入参', 'gui.pb.col_name': '名称', diff --git a/ui-tui/src/rpc/generated.ts b/ui-tui/src/rpc/generated.ts index e9d844a48..ffb8404a8 100644 --- a/ui-tui/src/rpc/generated.ts +++ b/ui-tui/src/rpc/generated.ts @@ -1893,6 +1893,27 @@ export interface PlaybookNodeShape { id: string; depends_on: string[]; } +/** + * One durable Harness alias and the registered agent behind it. + * + * This interface was referenced by `RavenRpcRoot`'s JSON-Schema + * via the `definition` "PlaybookWorkerShape". + */ +export interface PlaybookWorkerShape { + label: string; + agent: string; +} +/** + * The full worker detail; its brief is the durable per-job instruction. + * + * This interface was referenced by `RavenRpcRoot`'s JSON-Schema + * via the `definition` "PlaybookWorker". + */ +export interface PlaybookWorker { + label: string; + agent: string; + brief: string; +} /** * One playbook as the library list needs it. ``error`` is empty unless the * file would not parse, in which case it carries the reason and ``nodes`` is @@ -1908,6 +1929,9 @@ export interface PlaybookRow { name: string; description: string; task_summary: string; + schema_version: number; + artifact_kind: 'legacy' | 'workflow' | 'harness' | 'composite'; + workers: PlaybookWorkerShape[]; mode: 'dag' | 'prompt'; confirm: boolean; origin: string; @@ -1968,6 +1992,9 @@ export interface PlaybookDetail { description: string; task_summary: string; version: number; + schema_version: number; + artifact_kind: 'legacy' | 'workflow' | 'harness' | 'composite'; + workers: PlaybookWorker[]; mode: 'dag' | 'prompt'; confirm: boolean; origin: string; @@ -2361,6 +2388,7 @@ export interface SessionHistoryResult { export interface TurnSendParams { session_key: string; content: string; + playbook_mode?: 'off' | 'task' | 'persona'; channel?: string; chat_id?: string; sender_id?: string; diff --git a/ui-web/src/demo/154-playbooks.js b/ui-web/src/demo/154-playbooks.js index d2fa5f408..9568528d9 100644 --- a/ui-web/src/demo/154-playbooks.js +++ b/ui-web/src/demo/154-playbooks.js @@ -269,6 +269,9 @@ DS.playbooks ??= { name: p.name, description: p.description, task_summary: p.task_summary, + schema_version: 1, + artifact_kind: 'legacy', + workers: [], mode: p.mode, confirm: p.confirm, origin: p.origin, @@ -284,6 +287,9 @@ DS.playbooks ??= { description: p.description, task_summary: p.task_summary, version: p.version, + schema_version: 1, + artifact_kind: 'legacy', + workers: [], mode: p.mode, confirm: p.confirm, origin: p.origin, diff --git a/ui-web/src/features/playbooks/PlaybooksPage.test.tsx b/ui-web/src/features/playbooks/PlaybooksPage.test.tsx index ccd747963..a56dbdce2 100644 --- a/ui-web/src/features/playbooks/PlaybooksPage.test.tsx +++ b/ui-web/src/features/playbooks/PlaybooksPage.test.tsx @@ -25,6 +25,9 @@ function row(over: Partial & { name: string }): PlaybookRow { description: 'what it is for', task_summary: 'what running it dispatches', mode: 'dag', + schema_version: 1, + artifact_kind: 'legacy', + workers: [], confirm: true, origin: 'user', disabled: false, @@ -42,6 +45,9 @@ function detail(over: Partial & { name: string }): PlaybookDetai description: 'what it is for', task_summary: 'what running it dispatches', version: 1, + schema_version: 1, + artifact_kind: 'legacy', + workers: [], mode: 'dag', confirm: true, origin: 'user', @@ -115,6 +121,27 @@ describe('the playbook library', () => { expect(document.querySelector('.pbcard .pbcap')?.textContent).toBe('gui.pb.shape_live') }) + it('draws a Harness-only artifact as its reusable workers', async () => { + install({ + list: async () => [ + row({ + name: 'launch-crew', + schema_version: 2, + artifact_kind: 'harness', + mode: 'prompt', + nodes: [], + workers: [ + { label: 'signal-extractor', agent: 'Researcher' }, + { label: 'brief-writer', agent: 'Writer' } + ] + }) + ] + }) + await mount() + expect(document.querySelectorAll('.pbcard .pbcell.worker')).toHaveLength(2) + expect(document.querySelector('.pbcard .pbcap')?.textContent).toBe('gui.pb.shape_harness {"n":2}') + }) + it('counts the matches while a query is on, not the library', async () => { install({ list: async () => [ @@ -708,6 +735,50 @@ describe('the graph-level fields', () => { expect(screen.getByText('v3')).toBeTruthy() }) + it('shows the reusable Harness beside a composite Workflow', async () => { + await open({ + schema_version: 2, + artifact_kind: 'composite', + workers: [ + { + label: 'signal-extractor', + agent: 'Researcher', + brief: 'Extract only decision-relevant facts.' + }, + { + label: 'brief-writer', + agent: 'Writer', + brief: 'Turn verified facts into a concise executive brief.' + } + ] + }) + expect(screen.getByText('composite')).toBeTruthy() + expect(screen.getByText('signal-extractor')).toBeTruthy() + expect(screen.getByText('Researcher')).toBeTruthy() + expect(screen.getByText('Extract only decision-relevant facts.')).toBeTruthy() + expect(document.querySelector('.pbboard')).not.toBeNull() + }) + + it('does not invent a graph for a Harness-only artifact', async () => { + await open({ + schema_version: 2, + artifact_kind: 'harness', + mode: 'prompt', + nodes: [], + workers: [ + { + label: 'brand-voice', + agent: 'Writer', + brief: 'Speak in the saved brand voice.' + } + ] + }) + expect(tab('gui.pb.tab_harness').getAttribute('aria-selected')).toBe('true') + expect(document.querySelector('.pbboard')).toBeNull() + expect(screen.getByText('gui.pb.harness_only')).toBeTruthy() + expect(screen.getByText('brand-voice')).toBeTruthy() + }) + it('writes out both states of confirm', async () => { const confirmCell = (): string | undefined => [...document.querySelectorAll('.pbmeta div')] @@ -846,7 +917,6 @@ describe('the graph-level fields', () => { }) }) - describe('the credentials tab', () => { const carried = { params: { @@ -881,7 +951,7 @@ describe('the credentials tab', () => { } } const tab = (label: string): HTMLElement | undefined => - [...document.querySelectorAll('.pbtab')].find((b) => b.textContent === label) as HTMLElement | undefined + [...document.querySelectorAll('.pbtab')].find(b => b.textContent === label) as HTMLElement | undefined async function openCarried(over: Partial = {}) { const calls: string[] = [] @@ -933,11 +1003,11 @@ describe('the credentials tab', () => { fireEvent.click(tab('gui.pb.tab_credentials')!) await act(async () => {}) expect(tab('gui.pb.tab_credentials')!.getAttribute('aria-selected')).toBe('true') - const rows = [...document.querySelectorAll('.pbtbl tr .nm')].map((n) => n.textContent || '') - expect(rows.some((r) => r.startsWith('PROBE_TOKEN'))).toBe(true) - expect(rows.some((r) => r.startsWith('sentry'))).toBe(true) - expect(rows.some((r) => r.startsWith('topic'))).toBe(false) - expect(rows.some((r) => r.startsWith('tokened'))).toBe(false) + const rows = [...document.querySelectorAll('.pbtbl tr .nm')].map(n => n.textContent || '') + expect(rows.some(r => r.startsWith('PROBE_TOKEN'))).toBe(true) + expect(rows.some(r => r.startsWith('sentry'))).toBe(true) + expect(rows.some(r => r.startsWith('topic'))).toBe(false) + expect(rows.some(r => r.startsWith('tokened'))).toBe(false) /* The badge that says authorizing the host's sentry does nothing here. */ expect(document.body.textContent).toContain('gui.pb.cred_shadows_host') expect(document.body.textContent).toContain('gui.pb.cred_unset') @@ -963,7 +1033,7 @@ describe('the credentials tab', () => { const { calls } = await openCarried() fireEvent.click(tab('gui.pb.tab_credentials')!) await act(async () => {}) - const authorize = [...document.querySelectorAll('button')].find((b) => b.textContent === 'gui.pb.cred_authorize')! + const authorize = [...document.querySelectorAll('button')].find(b => b.textContent === 'gui.pb.cred_authorize')! fireEvent.click(authorize) await act(async () => {}) expect(calls).toEqual(['authorize sentry']) diff --git a/ui-web/src/features/playbooks/PlaybooksPage.tsx b/ui-web/src/features/playbooks/PlaybooksPage.tsx index 6dfbfe5bf..fee15df19 100644 --- a/ui-web/src/features/playbooks/PlaybooksPage.tsx +++ b/ui-web/src/features/playbooks/PlaybooksPage.tsx @@ -20,7 +20,13 @@ import { cardPlan, edge } from './shape' import * as store from './store' import type { Dims } from '../dag/graph' -import type { PlaybookCredentialParam, PlaybookCredentialServer, PlaybookDetail, PlaybookNode, PlaybookRow } from './types' +import type { + PlaybookCredentialParam, + PlaybookCredentialServer, + PlaybookDetail, + PlaybookNode, + PlaybookRow +} from './types' import type { JSX } from 'react' /* One geometry, always. The card's diagram has two metrics because a card @@ -40,6 +46,26 @@ function Tile({ name }: { name: string }): JSX.Element { /* ── the card's concept diagram ─────────────────────────────────────── */ function Concept({ row }: { row: PlaybookRow }): JSX.Element { + if (row.artifact_kind === 'harness') { + const workers = row.workers.slice(0, 4) + const width = workers.length * 48 + Math.max(0, workers.length - 1) * 16 + return ( + + ) + } if (row.mode === 'prompt') { /* No stored graph to draw: the shape is composed per run. Three dashed middle boxes say "a fan-out of some width", which is the only true thing @@ -54,7 +80,7 @@ function Concept({ row }: { row: PlaybookRow }): JSX.Element { aria-hidden="true" > - {[0, 1, 2].map((i) => { + {[0, 1, 2].map(i => { const y = i * 23 return ( @@ -71,7 +97,7 @@ function Concept({ row }: { row: PlaybookRow }): JSX.Element { ) } const plan = cardPlan(row.nodes) - const at = new Map(plan.cells.map((c) => [c.id, c])) + const at = new Map(plan.cells.map(c => [c.id, c])) const { W, H } = plan.metric return ( {t('gui.pb.shape_live')} : null} + {row.artifact_kind === 'harness' ? ( + {t('gui.pb.shape_harness', { n: row.workers.length })} + ) : row.mode === 'prompt' ? ( + {t('gui.pb.shape_live')} + ) : null} )} @@ -208,7 +238,7 @@ function Library(): JSX.Element { value={s.query} placeholder={t('gui.pb.search')} aria-label={t('gui.pb.search')} - onChange={(e) => store.search(e.target.value)} + onChange={e => store.search(e.target.value)} /> {/* While a query is on, the count is about the query -- a total nobody @@ -230,7 +260,7 @@ function Library(): JSX.Element {
{t(s.query ? 'gui.pb.none_match' : 'gui.pb.none')}
) : (
- {rows.map((r) => ( + {rows.map(r => ( ))}
@@ -273,21 +303,13 @@ function Prompt({ text }: { text: string }): JSX.Element { ) } -function Canvas({ - detail, - picked, - dims -}: { - detail: PlaybookDetail - picked: string | null - dims: Dims -}): JSX.Element { +function Canvas({ detail, picked, dims }: { detail: PlaybookDetail; picked: string | null; dims: Dims }): JSX.Element { const nodes = detail.nodes const { at, width, height } = layout(nodes, dims) /* Session groups: nodes sharing (subagent, instance). The only fact the arrows cannot carry -- an edge says "after", not "in the same session". */ const groups = new Map() - nodes.forEach((n) => { + nodes.forEach(n => { if (!n.instance) return const key = n.subagent + '@' + n.instance const g = groups.get(key) @@ -298,12 +320,12 @@ function Canvas({
{[...groups.entries()].map(([key, members]) => { if (members.length < 2) return null - const pts = members.map((m) => at.get(m.id)).filter(Boolean) as Array<{ x: number; y: number }> + const pts = members.map(m => at.get(m.id)).filter(Boolean) as Array<{ x: number; y: number }> if (pts.length < 2) return null - const x0 = Math.min(...pts.map((p) => p.x)) - 11 - const y0 = Math.min(...pts.map((p) => p.y)) - 11 - const x1 = Math.max(...pts.map((p) => p.x)) + dims.W + 11 - const y1 = Math.max(...pts.map((p) => p.y)) + dims.H + 11 + const x0 = Math.min(...pts.map(p => p.x)) - 11 + const y0 = Math.min(...pts.map(p => p.y)) - 11 + const x1 = Math.max(...pts.map(p => p.x)) + dims.W + 11 + const y1 = Math.max(...pts.map(p => p.y)) + dims.H + 11 return (
{t('gui.pb.lane', { handle: members[0]?.instance || '' })} @@ -311,8 +333,8 @@ function Canvas({ ) })} - {nodes.map((n) => { + {nodes.map(n => { const p = at.get(n.id) if (!p) return null const blank = !n.subagent || !n.prompt_template @@ -369,8 +391,12 @@ function Listed({ of, carried }: { of: Written; carried?: string[] }): JSX.Eleme const shipped = new Set(carried || []) return ( - {of.items.map((x) => ( - + {of.items.map(x => ( + {x} ))} @@ -526,7 +552,7 @@ function Board({ detail, picked }: { detail: PlaybookDetail; picked: string | nu window.clearInterval(slow) slow = 0 } - setPort((prev) => (prev.w === w && prev.h === h ? prev : { w, h })) + setPort(prev => (prev.w === w && prev.h === h ? prev : { w, h })) } measure() window.addEventListener('resize', measure) @@ -582,7 +608,7 @@ function Board({ detail, picked }: { detail: PlaybookDetail; picked: string | nu zoom from the closure makes all of them compute the same result -- the flick lands as one step and the canvas feels stuck. */ const zoomBy = (factor: number, about?: { x: number; y: number }): void => { - setView((prev) => { + setView(prev => { const cur = prev ?? { x: 0, y: 0, z: 1 } const z = clampZoom(cur.z * factor) /* Zoom about a point: the graph coordinate under it has to stay under it, @@ -607,7 +633,7 @@ function Board({ detail, picked }: { detail: PlaybookDetail; picked: string | nu const dy = e.clientY - d.y if (!d.moved && Math.abs(dx) + Math.abs(dy) < DRAG_SLOP) return d.moved = true - setView((prev) => ({ x: d.ox + dx, y: d.oy + dy, z: prev?.z ?? 1 })) + setView(prev => ({ x: d.ox + dx, y: d.oy + dy, z: prev?.z ?? 1 })) } const onUp = (e: React.PointerEvent): void => { const d = drag.current @@ -628,8 +654,7 @@ function Board({ detail, picked }: { detail: PlaybookDetail; picked: string | nu const r = el.getBoundingClientRect() /* deltaY is in lines or pages on some mice; normalise before reading it. */ const raw = e.deltaMode === 1 ? e.deltaY * 16 : e.deltaMode === 2 ? e.deltaY * (port.h || 1) : e.deltaY - const factor = - Math.abs(raw) >= MOUSE_NOTCH ? (raw < 0 ? ZOOM_STEP : 1 / ZOOM_STEP) : Math.exp(-raw * WHEEL_GAIN) + const factor = Math.abs(raw) >= MOUSE_NOTCH ? (raw < 0 ? ZOOM_STEP : 1 / ZOOM_STEP) : Math.exp(-raw * WHEEL_GAIN) zoomBy(factor, { x: e.clientX - r.left, y: e.clientY - r.top }) } el.addEventListener('wheel', onWheel, { passive: false }) @@ -643,10 +668,10 @@ function Board({ detail, picked }: { detail: PlaybookDetail; picked: string | nu '=': () => zoomBy(ZOOM_STEP), '-': () => zoomBy(1 / ZOOM_STEP), '0': () => setView(framed()), - ArrowUp: () => setView((prev) => ({ ...(prev ?? at), y: (prev ?? at).y + pan })), - ArrowDown: () => setView((prev) => ({ ...(prev ?? at), y: (prev ?? at).y - pan })), - ArrowLeft: () => setView((prev) => ({ ...(prev ?? at), x: (prev ?? at).x + pan })), - ArrowRight: () => setView((prev) => ({ ...(prev ?? at), x: (prev ?? at).x - pan })) + ArrowUp: () => setView(prev => ({ ...(prev ?? at), y: (prev ?? at).y + pan })), + ArrowDown: () => setView(prev => ({ ...(prev ?? at), y: (prev ?? at).y - pan })), + ArrowLeft: () => setView(prev => ({ ...(prev ?? at), x: (prev ?? at).x + pan })), + ArrowRight: () => setView(prev => ({ ...(prev ?? at), x: (prev ?? at).x - pan })) } /* Arrow keys walk the steps while a step is picked (see PlaybooksApp); they pan the canvas only when the canvas itself has focus and nothing is. */ @@ -720,7 +745,7 @@ function Params({ params }: { params: PlaybookDetail['params'] }): JSX.Element { - {keys.map((k) => { + {keys.map(k => { const p = params[k] if (!p) return null return ( @@ -753,7 +778,7 @@ function Contract({ detail }: { detail: PlaybookDetail }): JSX.Element {

{t('gui.pb.sec_keywords')}

- {detail.keywords.map((k) => ( + {detail.keywords.map(k => ( {k} @@ -795,7 +820,7 @@ function CarriedServers({ servers }: { servers: NonNullable - {names.map((name) => { + {names.map(name => { const s = servers[name] if (!s) return null const stdio = !!s.command @@ -845,11 +870,30 @@ function CarriedServers({ servers }: { servers: NonNullable +

{t('gui.pb.sec_harness')}

+
+ {workers.map(worker => ( +
+ {worker.label} + {worker.agent} + {worker.brief ?

{worker.brief}

: null} +
+ ))} +
+
+ ) +} + function Detail({ detail }: { detail: PlaybookDetail }): JSX.Element { const s = store.getState() - const node = detail.nodes.find((n) => n.id === s.pickedNode) || null + const node = detail.nodes.find(n => n.id === s.pickedNode) || null const onGraph = s.tab === 'graph' const onCreds = s.tab === 'credentials' + const hasWorkflow = detail.artifact_kind !== 'harness' return ( <>
@@ -881,16 +925,29 @@ function Detail({ detail }: { detail: PlaybookDetail }): JSX.Element { told from a page that did not say. */} {String(detail.confirm)}
+
+ {t('gui.pb.f_artifact')} + {detail.artifact_kind} +
+ +
- {/* Only where there is something to hold: a playbook with no secret @@ -906,12 +963,17 @@ function Detail({ detail }: { detail: PlaybookDetail }): JSX.Element { {/* Hidden rather than unmounted: the board holds the reader's own pan and zoom, and a look at the contract must not throw it away. */} -
- {detail.mode === 'dag' ? ( +
+ {hasWorkflow && detail.mode === 'dag' ? ( <> + ) : !hasWorkflow ? ( +
+ {t('gui.pb.sec_harness')} +

{t('gui.pb.harness_only')}

+
) : ( /* One column, because there is no second thing: a panel beside this saying the graph is composed per run was a note about the absence @@ -929,8 +991,8 @@ function Detail({ detail }: { detail: PlaybookDetail }): JSX.Element { } function hasCredentialSlots(detail: PlaybookDetail): boolean { - const secret = Object.values(detail.params).some((p) => p.type === 'secret') - const oauth = Object.values(detail.mcp_servers || {}).some((sv) => sv.auth === 'oauth') + const secret = Object.values(detail.params).some(p => p.type === 'secret') + const oauth = Object.values(detail.mcp_servers || {}).some(sv => sv.auth === 'oauth') return secret || oauth } @@ -944,7 +1006,7 @@ function Credentials({ detail, s }: { detail: PlaybookDetail; s: store.Playbooks /* Null both while the first read is in flight and when the engine has no credentials surface (loadCredentials leaves it null then). */ if (!creds) return

{s.credsLoading ? '…' : t('gui.pb.cred_none')}

- const oauthServers = creds.servers.filter((sv) => sv.auth === 'oauth') + const oauthServers = creds.servers.filter(sv => sv.auth === 'oauth') return ( <>

{t('gui.pb.cred_intro')}

@@ -953,7 +1015,7 @@ function Credentials({ detail, s }: { detail: PlaybookDetail; s: store.Playbooks

{t('gui.pb.cred_sec_params')}

- {creds.params.map((p) => ( + {creds.params.map(p => ( ))} @@ -965,7 +1027,7 @@ function Credentials({ detail, s }: { detail: PlaybookDetail; s: store.Playbooks

{t('gui.pb.cred_sec_servers')}

- {oauthServers.map((sv) => ( + {oauthServers.map(sv => ( { + onKeyDown={e => { if (e.key === 'Enter') save() }} /> @@ -1083,7 +1145,7 @@ export function PlaybooksApp(): JSX.Element { const target = e.target as HTMLElement | null if (target && /INPUT|TEXTAREA/.test(target.tagName)) return const nodes = (s.detail as PlaybookDetail).nodes - const i = nodes.findIndex((n) => n.id === s.pickedNode) + const i = nodes.findIndex(n => n.id === s.pickedNode) if (i < 0) return const next = nodes[e.key === 'ArrowRight' ? Math.min(i + 1, nodes.length - 1) : Math.max(i - 1, 0)] if (next) store.pick(next.id) diff --git a/ui-web/src/rpc/generated.ts b/ui-web/src/rpc/generated.ts index 1ad5b7bae..41e22494a 100644 --- a/ui-web/src/rpc/generated.ts +++ b/ui-web/src/rpc/generated.ts @@ -3,7 +3,7 @@ // Source of truth: rpc-schema/openrpc.json (OpenRPC 1.2.6). // Drift check: `npm run gen:check` (CI runs this; a stale file fails the build). // -// 179 methods, 97 component schemas. +// 179 methods, 99 component schemas. /* eslint-disable */ /** @@ -1507,6 +1507,21 @@ export interface PlaybookNodeShape { id: string; depends_on: string[]; } +/** + * One durable Harness alias and the registered agent behind it. + */ +export interface PlaybookWorkerShape { + label: string; + agent: string; +} +/** + * The full worker detail; its brief is the durable per-job instruction. + */ +export interface PlaybookWorker { + label: string; + agent: string; + brief: string; +} /** * One playbook as the library list needs it. ``error`` is empty unless the * file would not parse, in which case it carries the reason and ``nodes`` is @@ -1519,6 +1534,9 @@ export interface PlaybookRow { name: string; description: string; task_summary: string; + schema_version: number; + artifact_kind: 'legacy' | 'workflow' | 'harness' | 'composite'; + workers: PlaybookWorkerShape[]; mode: 'dag' | 'prompt'; confirm: boolean; origin: string; @@ -1570,6 +1588,9 @@ export interface PlaybookDetail { description: string; task_summary: string; version: number; + schema_version: number; + artifact_kind: 'legacy' | 'workflow' | 'harness' | 'composite'; + workers: PlaybookWorker[]; mode: 'dag' | 'prompt'; confirm: boolean; origin: string; @@ -1834,6 +1855,7 @@ export interface SessionHistoryResult { export interface TurnSendParams { session_key: string; content: string; + playbook_mode?: 'off' | 'task' | 'persona'; channel?: string; chat_id?: string; sender_id?: string; diff --git a/ui-web/src/styles/page.css b/ui-web/src/styles/page.css index 61d4aa714..aa57da767 100644 --- a/ui-web/src/styles/page.css +++ b/ui-web/src/styles/page.css @@ -5823,6 +5823,7 @@ body:has(.app[data-page="on"]) #deskHost { display: none; } /* Structure only. Colour on a card would be an encoding with no key on the card; the executor's name is in the detail, where it can be read. */ .pbcell { fill: var(--faint); opacity: .5; } +.pbcell.worker { fill: var(--sky); opacity: .58; } .pbcell.open { fill: none; stroke: var(--sky); stroke-width: 1.1; stroke-dasharray: 3 3; } .pbedge { stroke: color-mix(in oklab, var(--faint) 45%, transparent); stroke-width: 1.2; fill: none; } .pbedge.cut { stroke-dasharray: 3 3; } @@ -5854,12 +5855,28 @@ body:has(.app[data-page="on"]) #deskHost { display: none; } /* The standing facts: label over value, packed left. Three short values spread across equal columns would read as three unrelated things. */ .pbmeta { - display: grid; grid-template-columns: repeat(3, max-content); gap: 56px; + display: grid; grid-template-columns: repeat(4, max-content); gap: 56px; padding: 15px 0 16px; border-top: 1px solid var(--line); } /* Named `pblab` / `pbval` rather than the bare `lab` / `val` the page already has: those two carry an uppercase transform and a 9px left margin, which is what put every value out of line with its own label. */ +.pbharness { margin: 0 0 18px; } +.pbharness h2 { margin: 0 0 10px; font-size: var(--step-0); font-weight: 600; } +.pbworkers { display: flex; flex-wrap: wrap; gap: 8px; } +.pbworker { + display: grid; grid-template-columns: max-content max-content; align-items: baseline; gap: 5px 12px; + min-width: 220px; max-width: 380px; padding: 10px 12px; + border: 1px solid var(--line); border-radius: 10px; background: var(--surface); +} +.pbworkername { font-family: var(--mono); font-size: 12.5px; color: var(--text); } +.pbworkeragent { font-family: var(--mono); font-size: 11px; color: var(--faint); } +.pbworker p { + grid-column: 1 / -1; margin: 0; font-size: 12px; line-height: 1.45; color: var(--muted); +} +@media (max-width: 760px) { + .pbmeta { grid-template-columns: repeat(2, max-content); gap: 20px 42px; } +} .pblab { display: block; font-size: 11.5px; font-weight: 600; color: var(--muted); margin-bottom: 5px; } .pbval { font-size: 15px; color: var(--text); } .pbval.mono { font-family: var(--mono); font-size: 14px; }