Skip to content

Commit 00602df

Browse files
CuzyoungCopilot
andcommitted
feat(slow-update): add config-controlled gated / force-injected modes
Add optimizer.slow_update_gate_with_selection to control how epoch-boundary slow-update guidance is applied: - false (default): force-injected - inject guidance into current & best unconditionally (unchanged behavior). - true: gated - evaluate the slow-update candidate on the selection set and accept/reject via the same validation gate as step-level updates (logic follows the SkillReflection ablation). Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
1 parent 42e555d commit 00602df

5 files changed

Lines changed: 155 additions & 51 deletions

File tree

‎configs/_base_/default.yaml‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -66,6 +66,7 @@ optimizer:
6666
skill_update_mode: patch # patch / rewrite_from_suggestions / full_rewrite_minibatch
6767
use_slow_update: true
6868
slow_update_samples: 20
69+
slow_update_gate_with_selection: false
6970
longitudinal_pair_policy: mixed # mixed / changed / unchanged
7071
use_meta_skill: true
7172

‎skillopt/config.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -98,6 +98,7 @@
9898
"optimizer.meta_learning_rate": "meta_edit_budget",
9999
"optimizer.use_slow_update": "use_slow_update",
100100
"optimizer.slow_update_samples": "slow_update_samples",
101+
"optimizer.slow_update_gate_with_selection": "slow_update_gate_with_selection",
101102
"optimizer.longitudinal_pair_policy": "longitudinal_pair_policy",
102103
"optimizer.use_meta_skill": "use_meta_skill",
103104
"evaluation.use_gate": "use_gate",

‎skillopt/engine/trainer.py‎

Lines changed: 141 additions & 34 deletions
Original file line numberDiff line numberDiff line change
@@ -865,6 +865,15 @@ def _persist_runtime_state(last_completed_step: int) -> None:
865865
else ""
866866
)
867867
)
868+
slow_gate_with_selection = bool(
869+
cfg.get("slow_update_gate_with_selection", False)
870+
)
871+
print(
872+
" [slow update] acceptance="
873+
+ ("gated (selection-set validation)"
874+
if slow_gate_with_selection
875+
else "force-accept (unconditional)")
876+
)
868877
if current_score < 0:
869878
print(f"\n{'='*60}")
870879
print(" BASELINE — evaluate initial skill on Selection set (valid_seen)")
@@ -1468,17 +1477,27 @@ def _persist_runtime_state(last_completed_step: int) -> None:
14681477
epoch_comparison_pairs = None
14691478
if (
14701479
slow_saved.get("slow_update_content")
1471-
and slow_saved.get("action") in {
1472-
"accept", "accept_new_best", "force_accept",
1473-
}
14741480
and epoch >= 2
14751481
):
1476-
current_skill = replace_slow_update_field(
1477-
current_skill, slow_saved["slow_update_content"],
1478-
)
1479-
best_skill = replace_slow_update_field(
1480-
best_skill, slow_saved["slow_update_content"],
1481-
)
1482+
action = slow_saved.get("action")
1483+
if slow_gate_with_selection:
1484+
# Gated mode (follow SkillReflection): re-apply the
1485+
# guidance to current_skill only when it was accepted.
1486+
if action in {"accept", "accept_new_best"}:
1487+
current_skill = replace_slow_update_field(
1488+
current_skill,
1489+
slow_saved["slow_update_content"],
1490+
)
1491+
elif action in {
1492+
"accept", "accept_new_best", "force_accept",
1493+
}:
1494+
# Force-accept mode: re-apply to both current & best.
1495+
current_skill = replace_slow_update_field(
1496+
current_skill, slow_saved["slow_update_content"],
1497+
)
1498+
best_skill = replace_slow_update_field(
1499+
best_skill, slow_saved["slow_update_content"],
1500+
)
14821501
elif epoch == 1:
14831502
# Epoch 1: inject empty placeholder
14841503
os.makedirs(slow_dir, exist_ok=True)
@@ -1618,31 +1637,119 @@ def _persist_runtime_state(last_completed_step: int) -> None:
16181637
"observed across adjacent epochs."
16191638
)
16201639

1621-
# Slow update field is force-updated into both
1622-
# current_skill and best_skill unconditionally.
1623-
# The epoch-level longitudinal guidance should always
1624-
# persist — it must not be gated by step-level
1625-
# selection scores.
1626-
slow_content = slow_result["slow_update_content"]
1627-
current_skill = replace_slow_update_field(
1628-
current_skill, slow_content,
1629-
)
1630-
best_skill = replace_slow_update_field(
1631-
best_skill, slow_content,
1632-
)
1633-
# Update caches so downstream steps use the
1634-
# slow-update-injected skill for hashing.
1635-
slow_candidate_hash = skill_hash(current_skill)
1636-
sel_cache[slow_candidate_hash] = (current_score, 0.0)
1637-
1638-
slow_result["action"] = "force_accept"
1639-
current_origin = f"slow_update_epoch_{epoch:02d}"
1640-
1641-
print(
1642-
f" [slow update] force-injected into current & best "
1643-
f"({len(slow_content)} chars), "
1644-
f"{slow_time}s"
1645-
)
1640+
# Slow update acceptance — two modes selected via
1641+
# `optimizer.slow_update_gate_with_selection`.
1642+
if slow_gate_with_selection:
1643+
# ── Gated mode (follow SkillReflection) ──────────
1644+
# Evaluate the slow-update candidate on the
1645+
# selection set and accept/reject via the same
1646+
# validation gate used for step-level updates.
1647+
if slow_candidate_hash in sel_cache:
1648+
slow_sel_hard, slow_sel_soft = sel_cache[
1649+
slow_candidate_hash
1650+
]
1651+
print(
1652+
f" [slow gate] cache hit: "
1653+
f"hard={slow_sel_hard:.4f}"
1654+
)
1655+
else:
1656+
sel_env, sel_n = _build_eval_env(
1657+
split="valid_seen",
1658+
env_num=cfg["sel_env_num"],
1659+
seed=seed,
1660+
)
1661+
print(f" [slow gate] selection items={sel_n}")
1662+
slow_eval_dir = os.path.join(
1663+
slow_dir, "selection_eval",
1664+
)
1665+
slow_eval_results = adapter.rollout(
1666+
sel_env, slow_candidate, slow_eval_dir,
1667+
)
1668+
slow_sel_hard, slow_sel_soft = compute_score(
1669+
slow_eval_results
1670+
)
1671+
sel_cache[slow_candidate_hash] = (
1672+
slow_sel_hard, slow_sel_soft,
1673+
)
1674+
1675+
slow_gate = evaluate_gate(
1676+
candidate_skill=slow_candidate,
1677+
cand_hard=slow_sel_hard,
1678+
current_skill=current_skill,
1679+
current_score=current_score,
1680+
best_skill=best_skill,
1681+
best_score=best_score,
1682+
best_step=best_step,
1683+
global_step=global_step,
1684+
cand_soft=slow_sel_soft,
1685+
metric=gate_metric,
1686+
mixed_weight=gate_mixed_weight,
1687+
)
1688+
slow_result["selection_hard"] = slow_sel_hard
1689+
slow_result["selection_soft"] = slow_sel_soft
1690+
slow_result["action"] = slow_gate.action
1691+
prev_current = current_score
1692+
prev_best = best_score
1693+
current_skill = slow_gate.current_skill
1694+
current_score = slow_gate.current_score
1695+
best_skill = slow_gate.best_skill
1696+
best_score = slow_gate.best_score
1697+
best_step = slow_gate.best_step
1698+
if slow_gate.action in {"accept", "accept_new_best"}:
1699+
current_origin = (
1700+
f"slow_update_epoch_{epoch:02d}"
1701+
)
1702+
if slow_gate.action == "accept_new_best":
1703+
best_origin = current_origin
1704+
print(
1705+
f" [slow gate] ACCEPT (new best) "
1706+
f"hard={slow_sel_hard:.4f} > "
1707+
f"prev best {prev_best:.4f}"
1708+
)
1709+
elif slow_gate.action == "accept":
1710+
print(
1711+
f" [slow gate] ACCEPT "
1712+
f"hard={slow_sel_hard:.4f} > "
1713+
f"current={prev_current:.4f}"
1714+
)
1715+
else:
1716+
print(
1717+
f" [slow gate] REJECT "
1718+
f"hard={slow_sel_hard:.4f} <= "
1719+
f"current={current_score:.4f}"
1720+
)
1721+
print(
1722+
f" [slow update] guidance written "
1723+
f"({len(slow_result['slow_update_content'])} "
1724+
f"chars), {slow_time}s"
1725+
)
1726+
else:
1727+
# ── Force-accept mode (default) ──────────────────
1728+
# The epoch-level longitudinal guidance is injected
1729+
# into both current_skill and best_skill
1730+
# unconditionally — it must not be gated by
1731+
# step-level selection scores.
1732+
slow_content = slow_result["slow_update_content"]
1733+
current_skill = replace_slow_update_field(
1734+
current_skill, slow_content,
1735+
)
1736+
best_skill = replace_slow_update_field(
1737+
best_skill, slow_content,
1738+
)
1739+
# Update caches so downstream steps use the
1740+
# slow-update-injected skill for hashing.
1741+
slow_candidate_hash = skill_hash(current_skill)
1742+
sel_cache[slow_candidate_hash] = (current_score, 0.0)
1743+
1744+
slow_result["action"] = "force_accept"
1745+
current_origin = f"slow_update_epoch_{epoch:02d}"
1746+
1747+
print(
1748+
f" [slow update] force-injected into "
1749+
f"current & best "
1750+
f"({len(slow_content)} chars), "
1751+
f"{slow_time}s"
1752+
)
16461753
else:
16471754
slow_result = slow_result or {}
16481755
slow_result["action"] = "no_content"

‎skillopt/optimizer/meta_skill.py‎

Lines changed: 3 additions & 11 deletions
Original file line numberDiff line numberDiff line change
@@ -41,14 +41,6 @@ def run_meta_skill(
4141
"""Produce updated optimizer-side meta skill from adjacent epochs."""
4242
actual_system = system_prompt if system_prompt is not None else load_prompt("meta_skill")
4343

44-
prev_skill_display = prev_skill
45-
if len(prev_skill_display) > 6000:
46-
prev_skill_display = prev_skill_display[:6000] + "\n...[truncated]..."
47-
48-
curr_skill_display = curr_skill
49-
if len(curr_skill_display) > 6000:
50-
curr_skill_display = curr_skill_display[:6000] + "\n...[truncated]..."
51-
5244
prev_meta_section = (
5345
prev_meta_skill_content.strip()
5446
if prev_meta_skill_content and prev_meta_skill_content.strip()
@@ -57,8 +49,8 @@ def run_meta_skill(
5749

5850
comparison_text = format_comparison_text(comparison_pairs)
5951
user = (
60-
f"## Previous Epoch Last-Step Skill\n{prev_skill_display}\n\n"
61-
f"## Current Epoch Last-Step Skill\n{curr_skill_display}\n\n"
52+
f"## Previous Epoch Last-Step Skill\n{prev_skill}\n\n"
53+
f"## Current Epoch Last-Step Skill\n{curr_skill}\n\n"
6254
f"## Previous Optimizer Meta Skill\n"
6355
f"The following optimizer memory was available during the current epoch. "
6456
f"Reflect on whether it improved or harmed the quality of edits.\n\n"
@@ -71,7 +63,7 @@ def run_meta_skill(
7163
response, _ = chat_optimizer(
7264
system=actual_system,
7365
user=user,
74-
max_completion_tokens=3072,
66+
max_completion_tokens=16384,
7567
retries=3,
7668
stage="meta_skill",
7769
)

‎skillopt/optimizer/slow_update.py‎

Lines changed: 9 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -91,6 +91,11 @@ def replace_slow_update_field(skill: str, new_content: str) -> str:
9191
# ── Comparison text builder ─────────────────────────────────────────────────
9292

9393

94+
# NOTE: The character limits below (whole-trajectory cap + the per-field caps in
95+
# _read_trajectory and the comparison metadata) only trim the comparison samples
96+
# fed to the slow-update optimizer. They exist to cut token usage and speed up the
97+
# call; they do NOT affect what gets written into the skill. If you need richer
98+
# context for the longitudinal comparison, feel free to raise them.
9499
_MAX_TRAJ_CHARS = 3000
95100

96101

@@ -117,6 +122,8 @@ def _read_trajectory(rollout_dir: str, task_id: str) -> str:
117122
for entry in conversation:
118123
if not isinstance(entry, dict):
119124
continue
125+
# Per-field caps (cmd/obs/reasoning/etc.) keep each trajectory compact to
126+
# save tokens / time; raise them if you want fuller step detail.
120127
if entry.get("type") == "tool_call":
121128
cmd = _clip_text(entry.get("cmd"), 500)
122129
obs = _clip_text(entry.get("obs"), 800)
@@ -352,18 +359,14 @@ def run_slow_update(
352359
)
353360
comparison_text = format_comparison_text(pairs)
354361

355-
prev_skill_display = prev_skill
356-
if len(prev_skill_display) > 6000:
357-
prev_skill_display = prev_skill_display[:6000] + "\n...[truncated]..."
358-
359362
prev_guidance_section = (
360363
prev_slow_update_content.strip()
361364
if prev_slow_update_content and prev_slow_update_content.strip()
362365
else "(No previous guidance — this is the first slow update.)"
363366
)
364367

365368
user = (
366-
f"## Previous Epoch's Skill\n{prev_skill_display}\n\n"
369+
f"## Previous Epoch's Skill\n{prev_skill}\n\n"
367370
f"## Current Epoch's Skill\n{skill_content}\n\n"
368371
f"## Previous Slow Update Guidance\n"
369372
f"The following guidance was active during the current epoch. "
@@ -377,7 +380,7 @@ def run_slow_update(
377380
response, _ = chat_optimizer(
378381
system=actual_system,
379382
user=user,
380-
max_completion_tokens=4096,
383+
max_completion_tokens=16384,
381384
retries=3,
382385
stage="slow_update",
383386
)

0 commit comments

Comments
 (0)