From 4fabeeea082161a5edec9b471be05f847b4743c6 Mon Sep 17 00:00:00 2001 From: vam Date: Sun, 20 Sep 2026 12:38:30 +0800 Subject: [PATCH 1/2] Exclude Codex self-replay sessions from harvest --- skillopt_sleep/harvest_codex.py | 34 +++++++++++++++++++ tests/test_harvest_codex_replay.py | 53 ++++++++++++++++++++++++++++++ 2 files changed, 87 insertions(+) create mode 100644 tests/test_harvest_codex_replay.py diff --git a/skillopt_sleep/harvest_codex.py b/skillopt_sleep/harvest_codex.py index c50a237cb..a354abe0c 100644 --- a/skillopt_sleep/harvest_codex.py +++ b/skillopt_sleep/harvest_codex.py @@ -12,6 +12,7 @@ from skillopt_sleep.harvest import ( _detect_feedback, + _is_agent_session, _is_meta_prompt, _iter_jsonl, _project_matches, @@ -20,6 +21,37 @@ from skillopt_sleep.types import SessionDigest +_CODEX_REPLAY_PREFIXES = ( + "Complete the task. Apply the skill and memory rules exactly,", + "Complete the task. Apply the skill and memory rules EXACTLY,", +) +_CODEX_REPLAY_MARKERS = ( + "## CURRENT SKILL", + "## FAILED TASKS", + "## SUCCESSFUL TASKS", + "You are a strict grader", + "## TASK\n", + "## SKILL\n", + "## Skill\n", +) + + +def _is_codex_replay(digest: SessionDigest) -> bool: + """Detect the prompt shape emitted by SkillOpt's Codex replay backend.""" + if not digest.user_prompts: + return False + prompt = digest.user_prompts[0] + return ( + any(marker in prompt for marker in _CODEX_REPLAY_MARKERS) + or ( + any(prompt.startswith(prefix) for prefix in _CODEX_REPLAY_PREFIXES) + and "\n# Skill\n" in prompt + and "\n# Memory\n" in prompt + and "\n# Task\n" in prompt + ) + ) + + def _payload(rec: Dict[str, Any]) -> Dict[str, Any]: payload = rec.get("payload") return payload if isinstance(payload, dict) else {} @@ -225,6 +257,8 @@ def harvest_codex( continue if not _project_matches(digest.project or "", scope, invoked_project): continue + if _is_agent_session(digest) or _is_codex_replay(digest): + continue if since_iso and digest.ended_at and digest.ended_at < since_iso: continue digests.append(digest) diff --git a/tests/test_harvest_codex_replay.py b/tests/test_harvest_codex_replay.py new file mode 100644 index 000000000..cd2a6fed8 --- /dev/null +++ b/tests/test_harvest_codex_replay.py @@ -0,0 +1,53 @@ +"""Regression tests for excluding SkillOpt-generated Codex sessions.""" +from __future__ import annotations + +import json +import os +import tempfile +import unittest + +from skillopt_sleep.harvest_codex import harvest_codex + + +def _write_session(path: str, prompt: str) -> None: + with open(path, "w", encoding="utf-8") as handle: + for record in ( + { + "timestamp": "2026-09-20T00:00:00Z", + "payload": {"type": "turn_context", "cwd": "/repo"}, + }, + { + "timestamp": "2026-09-20T00:00:01Z", + "payload": {"type": "user_message", "message": prompt}, + }, + { + "timestamp": "2026-09-20T00:00:10Z", + "payload": {"type": "agent_message", "message": "Completed."}, + }, + ): + handle.write(json.dumps(record) + "\n") + + +class TestCodexReplayHarvest(unittest.TestCase): + def test_skillopt_replay_session_is_excluded(self): + prompt = ( + "Complete the task. Apply the skill and memory rules EXACTLY, including " + "any rule about searching before answering.\n\n" + "# Skill\nlearned rules\n\n# Memory\nprior notes\n\n" + "# Task\nanswer the task\n\nReturn ONLY the final answer." + ) + with tempfile.TemporaryDirectory() as tmp: + _write_session(os.path.join(tmp, "replay.jsonl"), prompt) + self.assertEqual(harvest_codex(tmp, scope="all"), []) + + def test_real_user_session_is_preserved(self): + with tempfile.TemporaryDirectory() as tmp: + _write_session(os.path.join(tmp, "real.jsonl"), "Please update the parser.") + digests = harvest_codex(tmp, scope="all") + + self.assertEqual(len(digests), 1) + self.assertEqual(digests[0].session_id, "real") + + +if __name__ == "__main__": + unittest.main() From 59be37e50a8f4069c8ecdf55ff6594b135479043 Mon Sep 17 00:00:00 2001 From: vam Date: Sat, 10 Oct 2026 17:26:53 +0800 Subject: [PATCH 2/2] fix: mark Codex engine sessions with explicit provenance --- docs/sleep/README.md | 8 +++++ skillopt_sleep/backend.py | 5 +++ skillopt_sleep/harvest_codex.py | 12 ++----- tests/test_codex_cli_prompt_stdin.py | 4 ++- tests/test_harvest_codex_replay.py | 47 ++++++++++++++++++++++++++++ 5 files changed, 65 insertions(+), 11 deletions(-) diff --git a/docs/sleep/README.md b/docs/sleep/README.md index f47556eeb..099a469f5 100644 --- a/docs/sleep/README.md +++ b/docs/sleep/README.md @@ -129,6 +129,14 @@ One engine, thin per-agent shells (see [`plugins/`](https://github.com/microsoft | **Devin** | [`plugins/devin`](https://github.com/microsoft/SkillOpt/tree/main/plugins/devin) | register `plugins/devin/mcp_server.py` as an MCP server | | **OpenClaw** | [`plugins/openclaw`](https://github.com/microsoft/SkillOpt/tree/main/plugins/openclaw) | adapt the reference wrapper and paths for your installation | +### Codex replay provenance + +Codex backend calls carry a `[skillopt-sleep:codex-engine:v1]` prefix so the +harvester can exclude engine-generated attempt, judge, reflection, and tool +sessions, including custom prompt templates. Ordinary user sessions that quote +headings such as `## TASK` or `## CURRENT SKILL` remain eligible for harvesting. +Legacy tool-replay sessions are recognized by their complete prompt structure. + ### VS Code GitHub Copilot Chat Use `--source copilot` to harvest local VS Code GitHub Copilot Chat sessions. diff --git a/skillopt_sleep/backend.py b/skillopt_sleep/backend.py index 99e41f418..ac52abc01 100644 --- a/skillopt_sleep/backend.py +++ b/skillopt_sleep/backend.py @@ -1445,6 +1445,9 @@ def _call_once(self, prompt: str, *, max_tokens: int = 1024) -> str: timeout/exception/empty-output (with last_call_error set). ``_call`` wraps this with retries so a transient failure is NOT silently scored 0.""" import tempfile + from skillopt_sleep.harvest_codex import CODEX_REPLAY_SENTINEL + + prompt = CODEX_REPLAY_SENTINEL + "\n\n" + prompt out_path = tempfile.NamedTemporaryFile( prefix="codex_last_", suffix=".txt", delete=False ).name @@ -1549,6 +1552,7 @@ def attempt_with_tools(self, task, skill, memory, tools): # `search` shim and let it run (workspace-write so the shim can log). import tempfile, shutil, stat work = tempfile.mkdtemp(prefix="skillopt_sleep_codextools_") + from skillopt_sleep.harvest_codex import CODEX_REPLAY_SENTINEL calllog = os.path.join(work, "_tool_calls.log") out_path = os.path.join(work, "_last.txt") tool_names = tools or ["search"] @@ -1604,6 +1608,7 @@ def attempt_with_tools(self, task, skill, memory, tools): ] if self.model: cmd += ["-m", self.model] + prompt = CODEX_REPLAY_SENTINEL + "\n\n" + prompt # Prompt via stdin (`codex exec -`): the Windows .CMD shim truncates argv at the first CR/LF. cmd += ["-"] self.last_call_error = "" diff --git a/skillopt_sleep/harvest_codex.py b/skillopt_sleep/harvest_codex.py index a354abe0c..f44f1f836 100644 --- a/skillopt_sleep/harvest_codex.py +++ b/skillopt_sleep/harvest_codex.py @@ -20,20 +20,12 @@ from skillopt_sleep.staging import _SECRET_PATTERNS from skillopt_sleep.types import SessionDigest +CODEX_REPLAY_SENTINEL = "[skillopt-sleep:codex-engine:v1]" _CODEX_REPLAY_PREFIXES = ( "Complete the task. Apply the skill and memory rules exactly,", "Complete the task. Apply the skill and memory rules EXACTLY,", ) -_CODEX_REPLAY_MARKERS = ( - "## CURRENT SKILL", - "## FAILED TASKS", - "## SUCCESSFUL TASKS", - "You are a strict grader", - "## TASK\n", - "## SKILL\n", - "## Skill\n", -) def _is_codex_replay(digest: SessionDigest) -> bool: @@ -42,7 +34,7 @@ def _is_codex_replay(digest: SessionDigest) -> bool: return False prompt = digest.user_prompts[0] return ( - any(marker in prompt for marker in _CODEX_REPLAY_MARKERS) + prompt.startswith(CODEX_REPLAY_SENTINEL + "\n\n") or ( any(prompt.startswith(prefix) for prefix in _CODEX_REPLAY_PREFIXES) and "\n# Skill\n" in prompt diff --git a/tests/test_codex_cli_prompt_stdin.py b/tests/test_codex_cli_prompt_stdin.py index 5ce55cb28..dc9f422df 100644 --- a/tests/test_codex_cli_prompt_stdin.py +++ b/tests/test_codex_cli_prompt_stdin.py @@ -59,7 +59,9 @@ def fake_run(cmd, **kwargs): self.assertEqual(out, "ok") self.assertEqual(len(calls), 1) cmd, kwargs = calls[0] - _assert_prompt_over_stdin(self, cmd, kwargs, prompt) + from skillopt_sleep.harvest_codex import CODEX_REPLAY_SENTINEL + + _assert_prompt_over_stdin(self, cmd, kwargs, CODEX_REPLAY_SENTINEL + "\n\n" + prompt) def test_attempt_with_tools_sends_multiline_prompt_via_stdin(self): from skillopt_sleep.backend import CodexCliBackend diff --git a/tests/test_harvest_codex_replay.py b/tests/test_harvest_codex_replay.py index cd2a6fed8..07a5eff95 100644 --- a/tests/test_harvest_codex_replay.py +++ b/tests/test_harvest_codex_replay.py @@ -5,8 +5,11 @@ import os import tempfile import unittest +from unittest import mock +from skillopt_sleep.backend import CodexCliBackend from skillopt_sleep.harvest_codex import harvest_codex +from skillopt_sleep.types import ReplayResult, TaskRecord def _write_session(path: str, prompt: str) -> None: @@ -29,6 +32,50 @@ def _write_session(path: str, prompt: str) -> None: class TestCodexReplayHarvest(unittest.TestCase): + def test_actual_backend_prompts_are_excluded(self): + prompts = [] + + def capture(cmd, **kwargs): + prompts.append(kwargs["input"]) + with open(cmd[cmd.index("-o") + 1], "w", encoding="utf-8") as handle: + if "You are SkillOpt's optimizer." in kwargs["input"]: + handle.write('[{"op": "add", "content": "Check the answer."}]') + else: + handle.write('{"score": 0.5, "reason": "synthetic"}') + return mock.Mock(returncode=0, stdout="", stderr="") + + backend = CodexCliBackend(codex_path="codex") + task = TaskRecord(id="t", project="/repo", intent="Answer the question", reference_kind="rubric") + with mock.patch("skillopt_sleep.backend.subprocess.run", side_effect=capture): + backend.attempt(task, "skill", "memory") + backend.judge(task, "answer") + backend.reflect( + [(task, ReplayResult(id="t", response="wrong", fail_reason="incorrect"))], + [], "skill", "memory", edit_budget=1, evolve_skill=True, evolve_memory=False, + ) + backend.attempt_with_tools(task, "skill", "memory", ["search"]) + # Benchmark/custom templates must carry the same provenance. + backend.attempt(TaskRecord(id="custom", project="/repo", intent="task", system="custom system"), "", "") + + self.assertEqual(len(prompts), 5) + for index, prompt in enumerate(prompts): + with self.subTest(operation=index), tempfile.TemporaryDirectory() as tmp: + _write_session(os.path.join(tmp, "engine.jsonl"), prompt) + self.assertEqual(harvest_codex(tmp), []) + + def test_quoted_headings_and_multiturn_user_sessions_are_preserved(self): + for heading in ("## CURRENT SKILL", "## TASK\n", "## FAILED TASKS", "You are a strict grader"): + for followup in (False, True): + with self.subTest(heading=heading, followup=followup), tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "user.jsonl") + _write_session(path, f"Please explain the {heading} section") + if followup: + with open(path, "a", encoding="utf-8") as handle: + handle.write(json.dumps({"payload": {"type": "user_message", "message": "perfect, thanks"}}) + "\n") + digests = harvest_codex(tmp) + self.assertEqual(len(digests), 1) + self.assertEqual(digests[0].n_user_turns, 2 if followup else 1) + def test_skillopt_replay_session_is_excluded(self): prompt = ( "Complete the task. Apply the skill and memory rules EXACTLY, including "