Skip to content

Commit 6da5ec1

Browse files
committed
test(sleep): cover every changed-document shape in staging (LG-40917)
1 parent d115799 commit 6da5ec1

1 file changed

Lines changed: 63 additions & 14 deletions

File tree

‎tests/test_sleep_engine.py‎

Lines changed: 63 additions & 14 deletions
Original file line numberDiff line numberDiff line change
@@ -1473,15 +1473,17 @@ def test_cycle_pins_the_exact_managed_skill_and_memory_bytes_it_read(self):
14731473
os.path.realpath(memory_path),
14741474
)
14751475

1476-
def test_cycle_stages_only_documents_that_changed(self):
1476+
def _assert_only_changed_documents_are_staged(
1477+
self, new_skill, new_memory, expect_skill, expect_memory
1478+
):
14771479
from skillopt_sleep.consolidate import ConsolidationResult
14781480

1481+
skill = "# managed baseline\nrule\n"
1482+
memory = "# memory baseline\npreference\n"
14791483
with tempfile.TemporaryDirectory() as proj, tempfile.TemporaryDirectory() as home:
14801484
target = os.path.join(proj, ".agents", "skills", "taste", "SKILL.md")
14811485
memory_path = os.path.join(proj, "CLAUDE.md")
14821486
os.makedirs(os.path.dirname(target), exist_ok=True)
1483-
skill = "# managed baseline\nrule\n"
1484-
memory = "# memory baseline\npreference\n"
14851487
with open(target, "w", encoding="utf-8") as handle:
14861488
handle.write(skill)
14871489
with open(memory_path, "w", encoding="utf-8") as handle:
@@ -1494,16 +1496,19 @@ def test_cycle_stages_only_documents_that_changed(self):
14941496
target_skill_path=target,
14951497
auto_adopt=False,
14961498
)
1499+
applied = []
1500+
if new_skill != skill:
1501+
applied.append(EditRecord("skill", "add", "sharpened rule"))
1502+
if new_memory != memory:
1503+
applied.append(EditRecord("memory", "add", "learned preference"))
14971504
result = ConsolidationResult(
14981505
accepted=True,
14991506
gate_action="accept_new_best",
15001507
baseline_score=0.1,
15011508
candidate_score=0.2,
1502-
new_skill=skill,
1503-
new_memory=memory + "learned preference\n",
1504-
applied_edits=[
1505-
EditRecord("memory", "add", "learned preference")
1506-
],
1509+
new_skill=new_skill,
1510+
new_memory=new_memory,
1511+
applied_edits=applied,
15071512
rejected_edits=[],
15081513
holdout_baseline=0.1,
15091514
holdout_candidate=0.2,
@@ -1518,20 +1523,64 @@ def test_cycle_stages_only_documents_that_changed(self):
15181523
):
15191524
outcome = run_sleep_cycle(cfg, seed_tasks=tasks)
15201525

1526+
# Staging never edits the live documents; adoption stays explicit.
1527+
with open(target, encoding="utf-8") as handle:
1528+
self.assertEqual(handle.read(), skill)
1529+
with open(memory_path, encoding="utf-8") as handle:
1530+
self.assertEqual(handle.read(), memory)
1531+
15211532
with open(
15221533
os.path.join(outcome.staging_dir, "manifest.json"),
15231534
encoding="utf-8",
15241535
) as handle:
15251536
manifest = json.load(handle)
1526-
self.assertFalse(manifest["has_managed_skill"])
1527-
self.assertTrue(manifest["has_managed_memory"])
1528-
self.assertFalse(
1529-
os.path.exists(os.path.join(outcome.staging_dir, "proposed_SKILL.md"))
1537+
# Manifest flags and artifact presence have to agree; a flag without
1538+
# its file (or a file without its flag) would break adoption.
1539+
self.assertEqual(manifest["has_managed_skill"], expect_skill)
1540+
self.assertEqual(manifest["has_managed_memory"], expect_memory)
1541+
self.assertEqual(
1542+
os.path.exists(
1543+
os.path.join(outcome.staging_dir, "proposed_SKILL.md")
1544+
),
1545+
expect_skill,
15301546
)
1531-
self.assertTrue(
1532-
os.path.exists(os.path.join(outcome.staging_dir, "proposed_CLAUDE.md"))
1547+
self.assertEqual(
1548+
os.path.exists(
1549+
os.path.join(outcome.staging_dir, "proposed_CLAUDE.md")
1550+
),
1551+
expect_memory,
15331552
)
15341553

1554+
def test_cycle_stages_only_documents_that_changed(self):
1555+
# The staging contract is byte/text equality, not semantic or whitespace
1556+
# normalized comparison: an accepted cycle proposes a document only when it
1557+
# actually rewrote it. Covered for every shape an accepted result can take,
1558+
# so a symmetric regression on the skill side cannot hide behind the
1559+
# memory-only case.
1560+
skill = "# managed baseline\nrule\n"
1561+
memory = "# memory baseline\npreference\n"
1562+
new_skill = skill + "prefer the shortest reproduction\n"
1563+
new_memory = memory + "learned preference\n"
1564+
cases = (
1565+
("neither_changed", skill, memory, False, False),
1566+
("skill_only", new_skill, memory, True, False),
1567+
("memory_only", skill, new_memory, False, True),
1568+
("both_changed", new_skill, new_memory, True, True),
1569+
# Whitespace-only is a real change under a byte-equality contract, so it
1570+
# is a positive case. If this ever fails, the comparison has started
1571+
# normalizing and the documented contract has silently moved.
1572+
("whitespace_only_skill", skill + "\n", memory, True, False),
1573+
("whitespace_only_memory", skill, memory + " \n", False, True),
1574+
)
1575+
for name, candidate_skill, candidate_memory, expect_skill, expect_memory in cases:
1576+
with self.subTest(case=name):
1577+
self._assert_only_changed_documents_are_staged(
1578+
candidate_skill,
1579+
candidate_memory,
1580+
expect_skill,
1581+
expect_memory,
1582+
)
1583+
15351584
def test_managed_skill_change_during_consolidation_refuses_the_night(self):
15361585
from skillopt_sleep.consolidate import ConsolidationResult
15371586
from skillopt_sleep.staging import StagingError, latest_staging

0 commit comments

Comments
 (0)