Skip to content
15 changes: 11 additions & 4 deletions agents/raven-code/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -477,12 +477,19 @@ def render_config(source: Path, partition: Path, mode: str | None = None, *, una
# a reply and exits, so the gate refuses every write and every command
# instead of prompting (measured 2026-09-08: the model could not edit
# one line and reported the task incomplete), and trunk's own one-shot
# spine names this the operator's call. The ACP hosting keeps the
# default: raven dispatching a sub-agent answers those prompts itself,
# and a person in an editor should still be asked. Builtin refusals
# (the catastrophic-command list) hold in every mode, and an explicit
# spine names this the operator's call. Builtin refusals (the
# catastrophic-command list) hold in every mode, and an explicit
# permissions block in a custom config wins.
config.setdefault("permissions", {}).setdefault("mode", "full")
else:
# The ACP hosting used to inherit trunk's default because that default
# was the ask tier; it has since moved to smart, where a reviewer
# speaks for that tier and lets most of it through. That is a product
# decision about raven's own surfaces, and this is not one of them:
# the person here is in an editor, watching an agent work on their
# checkout, and the prompt is how they see each write before it lands.
# Pinned rather than inherited so the tier stops moving under them.
config.setdefault("permissions", {}).setdefault("mode", "ask")

# Declared, not merged: the engine composes a profile per session from the
# catalogue over session/set_mode, and --mode only picks the starting entry.
Expand Down
9 changes: 8 additions & 1 deletion raven/config/schema.py
Original file line number Diff line number Diff line change
Expand Up @@ -1268,9 +1268,16 @@ class PermissionsConfig(Base):
(``"git *"``) each mapping to a tier; several matching patterns resolve to
the strictest. ``judge_model`` pins the smart-mode reviewer to one model id;
empty means the running turn's own binding.

``smart`` out of the box. ``ask`` stopped the agent on every mutation of a
conversation, which a reader answers by reflex rather than by reading, and
a prompt answered by reflex is not a gate. Smart is not the weaker setting
it sounds like: builtin denials and user deny rules hold in every mode, the
reviewer speaks only for the ask tier, and a reviewer that cannot run
leaves the call at the same prompt ``ask`` would have shown.
"""

mode: Literal["ask", "smart", "full"] = "ask"
mode: Literal["ask", "smart", "full"] = "smart"
Comment thread
gloryfromca marked this conversation as resolved.
tools: dict[str, str | dict[str, str]] = Field(default_factory=dict)
judge_model: str = ""
judge_timeout_seconds: float = 10.0
Expand Down
4 changes: 3 additions & 1 deletion raven/rpc/methods/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -54,7 +54,9 @@
"tui.theme": "default",
"tui.show_token_usage": True,
"language": "en",
"permissions.mode": "ask",
# What config.get reports as the default and what config.unset restores,
# so it has to be the value PermissionsConfig.mode carries.
"permissions.mode": "smart",
}


Expand Down
67 changes: 22 additions & 45 deletions raven/rpc/methods/console.py
Original file line number Diff line number Diff line change
Expand Up @@ -1048,13 +1048,13 @@ def _usage_range(params: dict):
async def settings_usage(params: dict, *, agent_loop_factory=None) -> dict:
"""Aggregate API usage for the settings page.

LLM side reads the UsageTracker telemetry files
(``~/.raven/telemetry/usage-YYYY-MM-DD.jsonl``, one JSON row per call);
tool side counts ``tool_calls`` entries across session transcripts whose
file mtime falls inside the window. Both scans are read-only and bounded
by ``days`` (default 30, max 90).
Both halves read the same UsageTracker telemetry files
(``~/.raven/telemetry/usage-YYYY-MM-DD.jsonl``, one JSON row per call, one
per tool call), so one range means one thing across the whole reply. The
scan is read-only and bounded by ``days`` (default 30, max 90). Session
transcripts are read for their titles only.
"""
from datetime import datetime, timedelta
from datetime import datetime

from raven.config.loader import load_config

Expand All @@ -1081,7 +1081,6 @@ def empty_totals() -> dict[str, Any]:

selected_session = params.get("session_key") or None
sessions: set[str] = set()
member_sessions: set[str] = {selected_session} if selected_session else set()
models: dict[str, dict[str, Any]] = {}
total = empty_totals()
# The same resolution the writer uses (usage_tracker._default_telemetry_dir):
Expand Down Expand Up @@ -1114,8 +1113,6 @@ def empty_totals() -> dict[str, Any]:
sessions.add(root)
if selected_session and root != selected_session:
continue
if isinstance(row.get("session_key"), str):
member_sessions.add(row["session_key"])
name = str(row.get("model") or "?")
acc = models.setdefault(name, {"model": name, **empty_totals()})
cost = reported_cost(row.get("cost_usd")) if row.get("schema_version") == 2 else None
Expand All @@ -1138,7 +1135,6 @@ def empty_totals() -> dict[str, Any]:
session_titles: dict[str, str] = {}
tools: dict[str, int] = {}
tool_total = 0
telemetry_tool_ids: set[str] = set()
try:
for day in dates:
p = tel_dir / f"usage-{day.isoformat()}.jsonl"
Expand All @@ -1153,60 +1149,41 @@ def empty_totals() -> dict[str, Any]:
if selected_session and root != selected_session:
continue
name = row.get("name")
if isinstance(row.get("tool_call_id"), str):
telemetry_tool_ids.add(row["tool_call_id"])
if isinstance(name, str):
tools[name] = tools.get(name, 0) + 1
tool_total += 1
except Exception:
continue
# Titles only. Tool calls were also counted from transcripts here, to
# cover conversations older than the day tool rows started being
# written, and a transcript counted as in-range when its file mtime
# was -- which gave the range every tool call the conversation had ever
# made while its model calls, dated per day, stayed outside it. A reply
# cannot carry two readings of one range: a page showing thousands of
# tool calls beside no model calls reads as broken, and is.
sess_root = Path(load_config().workspace_path) / "sessions"
# Both ends, because the range is a window rather than a floor: the
# daily telemetry files this falls back for are read for the selected
# days only, so a transcript touched after `to` would add tool calls
# the other two tallies of the same reply do not have.
# A floor rather than a window: a transcript last written before the
# range cannot name a session the range saw, and which sessions it saw
# is what the telemetry above already answered.
cutoff = datetime.combine(frm, datetime.min.time()).timestamp()
until = datetime.combine(to + timedelta(days=1), datetime.min.time()).timestamp()
for p in sess_root.glob("*/*.jsonl"):
try:
mtime = p.stat().st_mtime
if mtime < cutoff or mtime >= until:
if p.stat().st_mtime < cutoff:
continue
lines = p.read_text(encoding="utf-8").splitlines()
except Exception:
continue
metadata = None
for line in lines:
try:
entry = json.loads(line)
except ValueError:
continue
if isinstance(entry, dict) and entry.get("_type") == "metadata":
metadata = entry
if metadata:
key = metadata.get("key")
title = (metadata.get("metadata") or {}).get("title")
if isinstance(key, str) and isinstance(title, str):
session_titles[key] = title
if selected_session and (not metadata or metadata.get("key") not in member_sessions):
continue
for line in lines:
try:
msg = json.loads(line)
except ValueError:
if not isinstance(entry, dict) or entry.get("_type") != "metadata":
continue
if not isinstance(msg, dict):
continue
for tc in msg.get("tool_calls") or []:
if not isinstance(tc, dict):
continue
name = tc.get("name") or (tc.get("function") or {}).get("name")
call_id = tc.get("id")
if isinstance(call_id, str) and call_id in telemetry_tool_ids:
continue
if name:
tools[str(name)] = tools.get(str(name), 0) + 1
tool_total += 1
key = entry.get("key")
title = (entry.get("metadata") or {}).get("title")
if isinstance(key, str) and key in sessions and isinstance(title, str):
session_titles[key] = title
except Exception:
logger.exception("settings.usage: tool scan failed")

Expand Down
24 changes: 14 additions & 10 deletions raven/rpc/methods/session.py
Original file line number Diff line number Diff line change
Expand Up @@ -839,19 +839,23 @@ async def session_archive(
agent_loop = _safe_invoke_factory(agent_loop_factory)
config = load_config()
mgr = manager_for(agent_loop, config)
session = mgr.get_or_create(session_key)
# Restore writes an explicit False rather than dropping the key: the auto
# archive pass in session.list skips any session that carries the key, so
# a restored session stays out of its reach for good.
session.metadata["archived"] = archived
if mgr.exists(session_key):
try:
mgr.save(session)
except Exception:
logger.warning("session.archive: failed to persist archive state for {}", session_key)
return {"archived": archived, "session_key": session_key, "pending": True}
return {"archived": archived, "session_key": session_key, "pending": False}
return {"archived": archived, "session_key": session_key, "pending": True}
#
# One appended key rather than a saved session, the same way the auto
# archive pass writes it: ``save`` rewrites the whole metadata record from
# this process's copy of it, so a second client holding the conversation
# from before the archive dropped the flag on its next save of anything --
# a title, a model -- and the conversation came back.
try:
persisted = mgr.append_metadata_patch(session_key, {"archived": archived})
Comment thread
gloryfromca marked this conversation as resolved.
except Exception:
logger.warning("session.archive: failed to persist archive state for {}", session_key)
persisted = False
if not persisted:
mgr.get_or_create(session_key).metadata["archived"] = archived
return {"archived": archived, "session_key": session_key, "pending": not persisted}


async def session_clear(
Expand Down
16 changes: 12 additions & 4 deletions raven/session/manager.py
Original file line number Diff line number Diff line change
Expand Up @@ -655,24 +655,32 @@ def _load(self, key: str) -> Session | None:
logger.warning("Failed to load session {}: {}", key, e)
return None

def append_metadata_patch(self, key: str, patch: dict[str, Any]) -> None:
def append_metadata_patch(self, key: str, patch: dict[str, Any]) -> bool:
"""Fold ``patch`` into a session's metadata by appending one record.

The last metadata record wins on load, so appending is enough -- and
the transcript is never read, which is what lets a housekeeping pass
touch hundreds of sessions without loading any of them.
touch hundreds of sessions without loading any of them. Merging into
the record on disk is also what keeps two clients from undoing each
other: whoever writes second keeps the other's keys, which a whole
``save`` of one client's copy of the metadata cannot do.

False when there was nothing to append to -- no transcript, or one with
no metadata record -- which a caller reporting whether a flag reached
the disk has to tell apart from a write that happened.
"""
path = self.session_path(key)
if not path.is_file():
return
return False
last, _count, _last_ts, _first, _preview = self._scan_file(path)
if last is None:
return
return False
merged = {**(last.get("metadata") or {}), **patch}
locked_append(path, [json.dumps({**last, "metadata": merged}, ensure_ascii=False)])
cached = self._cache.get(key)
if cached is not None:
cached.metadata.update(patch)
return True

def save(self, session: Session) -> None:
"""Save a session to disk.
Expand Down
18 changes: 10 additions & 8 deletions tests/test_agents_code_launcher.py
Original file line number Diff line number Diff line change
Expand Up @@ -953,14 +953,16 @@ def test_the_products_tool_face_is_the_forks_config_intent_minus_the_ledger(grou
# --- the permission gate: only the hosting with nobody to ask opens it ------------


def test_the_acp_render_leaves_the_ask_tier_alone(grounded):
"""Trunk's permission gate (permissions.mode, default ``ask``) prompts a
person before a write or a command, and refuses outright when the turn is
not interactive. The ACP hosting keeps that default on purpose: raven
dispatching a sub-agent answers those prompts itself
(``raven/acp_client/permissions.py`` approves every one), and a person in
an editor should still be asked -- opening the tier product-wide would
take their prompt away for good."""
def test_the_acp_render_pins_the_ask_tier(grounded):
"""The ACP hosting asks, whatever tier trunk defaults to.

It used to inherit that default, which was the ask tier. The default has
since moved to smart, where a reviewer speaks for the ask tier and lets
most of it through -- a product decision about raven's own surfaces. This
is not one of them: the person is in an editor watching an agent work on
their checkout, and the prompt is how they see each write before it lands.
Raven dispatching a sub-agent is unaffected either way, since
``raven/acp_client/permissions.py`` answers every prompt itself."""
from raven.config.loader import load_config

config = load_config(_render(grounded))
Expand Down
2 changes: 1 addition & 1 deletion tests/test_config_live.py
Original file line number Diff line number Diff line change
Expand Up @@ -422,7 +422,7 @@ def test_permissions_node_invalid_from_the_start_answers_defaults(tmp_path):
path = tmp_path / "config.json"
path.write_text('{"permissions": {"mode": "godmode"}}')
fresh = permissions_config(LiveConfig(path))
assert fresh.mode == "ask"
assert fresh.mode == "smart"
assert fresh.tools == {}


Expand Down
4 changes: 2 additions & 2 deletions tests/test_rpc_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -1085,9 +1085,9 @@ async def test_a_conversation_mode_stays_in_memory_and_off_the_default(fake_home
own = await config_get({"keys": ["permissions.mode"], "session_id": "s-1"})
assert own["config"]["permissions.mode"] == "full"
other = await config_get({"keys": ["permissions.mode"], "session_id": "s-2"})
assert other["config"]["permissions.mode"] == "ask"
assert other["config"]["permissions.mode"] == "smart"
default = await config_get({"keys": ["permissions.mode"]})
assert default["config"]["permissions.mode"] == "ask"
assert default["config"]["permissions.mode"] == "smart"


async def test_a_conversation_mode_moves_both_ways(fake_home: Path, own_mode) -> None:
Expand Down
87 changes: 87 additions & 0 deletions tests/test_rpc_session.py
Original file line number Diff line number Diff line change
Expand Up @@ -1831,6 +1831,93 @@ async def test_session_archive_persists_and_filters_the_list(tmp_path: Path, mon
assert [row["id"] for row in (await session_list({}))["sessions"]] == [session_key]


async def test_archiving_keeps_a_key_another_writer_added(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""Archiving speaks for the archived flag and for nothing else.

It used to persist by saving the whole session, which rewrites the metadata
record from this manager's copy of it -- so every key written to the file
after that copy was loaded was dropped by an unrelated archive. A page and
a terminal over one home are two managers over one file, and that is how a
conversation came back after being archived: not because archiving failed,
but because somebody else's save spoke for a flag it had never seen.
"""
cfg = load_config()
cfg.agents.defaults.workspace = str(tmp_path)
monkeypatch.setattr(session_module, "load_config", lambda: cfg)

session_key = "tui:20260610_100000_foreignkey"
mgr = SessionManager(tmp_path)
session = mgr.get_or_create(session_key)
session.add_message("user", "two writers hold this")
mgr.save(session)
monkeypatch.setattr("raven.session.resolve.build_manager", lambda cfg: mgr)

# Somebody else writes a flag straight to the file; this manager's copy of
# the metadata knows nothing about it.
SessionManager(tmp_path).append_metadata_patch(session_key, {"pinned": True})
assert mgr.get_or_create(session_key).metadata.get("pinned") is None

result = await session_archive({"session_id": session_key, "archived": True})
assert result == {"archived": True, "session_key": session_key, "pending": False}

reloaded = SessionManager(tmp_path).peek(session_key)
assert reloaded is not None
assert reloaded.metadata.get("archived") is True
assert reloaded.metadata.get("pinned") is True


async def test_archiving_a_session_with_no_transcript_says_so(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""Nothing reached the disk, and the reply says it rather than implying it.

A conversation minted but never saved has no file to fold the flag into.
The call answers ``pending``, and a client that reads that as a success
takes the row off its list and finds it back on the next load.
"""
cfg = load_config()
cfg.agents.defaults.workspace = str(tmp_path)
monkeypatch.setattr(session_module, "load_config", lambda: cfg)
mgr = SessionManager(tmp_path)
monkeypatch.setattr("raven.session.resolve.build_manager", lambda cfg: mgr)

session_key = "tui:20260610_100000_nofile"
result = await session_archive({"session_id": session_key, "archived": True})
assert result == {"archived": True, "session_key": session_key, "pending": True}
assert not mgr.exists(session_key)
assert mgr.get_or_create(session_key).metadata.get("archived") is True


async def test_a_refused_write_is_reported_rather_than_swallowed(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""A filesystem that refuses the append answers pending, not success.

The flag is still applied in memory so the session behaves as asked for as
long as this process lives, but the reply says it did not reach the disk --
a client that takes the row off its list on a plain success would find it
back on the next load.
"""
cfg = load_config()
cfg.agents.defaults.workspace = str(tmp_path)
monkeypatch.setattr(session_module, "load_config", lambda: cfg)

session_key = "tui:20260610_100000_refused"
mgr = SessionManager(tmp_path)
session = mgr.get_or_create(session_key)
session.add_message("user", "hello")
mgr.save(session)
monkeypatch.setattr("raven.session.resolve.build_manager", lambda cfg: mgr)

def refuse(*_args: object, **_kwargs: object) -> bool:
raise OSError("read-only file system")

monkeypatch.setattr(mgr, "append_metadata_patch", refuse)

result = await session_archive({"session_id": session_key, "archived": True})
assert result == {"archived": True, "session_key": session_key, "pending": True}
assert mgr.get_or_create(session_key).metadata["archived"] is True
assert SessionManager(tmp_path).peek(session_key).metadata.get("archived") is None


async def test_session_archive_via_dispatcher(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
cfg = load_config()
cfg.agents.defaults.workspace = str(tmp_path)
Expand Down
Loading
Loading