@@ -865,6 +865,15 @@ def _persist_runtime_state(last_completed_step: int) -> None:
865865 else ""
866866 )
867867 )
868+ slow_gate_with_selection = bool (
869+ cfg .get ("slow_update_gate_with_selection" , False )
870+ )
871+ print (
872+ " [slow update] acceptance="
873+ + ("gated (selection-set validation)"
874+ if slow_gate_with_selection
875+ else "force-accept (unconditional)" )
876+ )
868877 if current_score < 0 :
869878 print (f"\n { '=' * 60 } " )
870879 print (" BASELINE — evaluate initial skill on Selection set (valid_seen)" )
@@ -1468,17 +1477,27 @@ def _persist_runtime_state(last_completed_step: int) -> None:
14681477 epoch_comparison_pairs = None
14691478 if (
14701479 slow_saved .get ("slow_update_content" )
1471- and slow_saved .get ("action" ) in {
1472- "accept" , "accept_new_best" , "force_accept" ,
1473- }
14741480 and epoch >= 2
14751481 ):
1476- current_skill = replace_slow_update_field (
1477- current_skill , slow_saved ["slow_update_content" ],
1478- )
1479- best_skill = replace_slow_update_field (
1480- best_skill , slow_saved ["slow_update_content" ],
1481- )
1482+ action = slow_saved .get ("action" )
1483+ if slow_gate_with_selection :
1484+ # Gated mode (follow SkillReflection): re-apply the
1485+ # guidance to current_skill only when it was accepted.
1486+ if action in {"accept" , "accept_new_best" }:
1487+ current_skill = replace_slow_update_field (
1488+ current_skill ,
1489+ slow_saved ["slow_update_content" ],
1490+ )
1491+ elif action in {
1492+ "accept" , "accept_new_best" , "force_accept" ,
1493+ }:
1494+ # Force-accept mode: re-apply to both current & best.
1495+ current_skill = replace_slow_update_field (
1496+ current_skill , slow_saved ["slow_update_content" ],
1497+ )
1498+ best_skill = replace_slow_update_field (
1499+ best_skill , slow_saved ["slow_update_content" ],
1500+ )
14821501 elif epoch == 1 :
14831502 # Epoch 1: inject empty placeholder
14841503 os .makedirs (slow_dir , exist_ok = True )
@@ -1618,31 +1637,119 @@ def _persist_runtime_state(last_completed_step: int) -> None:
16181637 "observed across adjacent epochs."
16191638 )
16201639
1621- # Slow update field is force-updated into both
1622- # current_skill and best_skill unconditionally.
1623- # The epoch-level longitudinal guidance should always
1624- # persist — it must not be gated by step-level
1625- # selection scores.
1626- slow_content = slow_result ["slow_update_content" ]
1627- current_skill = replace_slow_update_field (
1628- current_skill , slow_content ,
1629- )
1630- best_skill = replace_slow_update_field (
1631- best_skill , slow_content ,
1632- )
1633- # Update caches so downstream steps use the
1634- # slow-update-injected skill for hashing.
1635- slow_candidate_hash = skill_hash (current_skill )
1636- sel_cache [slow_candidate_hash ] = (current_score , 0.0 )
1637-
1638- slow_result ["action" ] = "force_accept"
1639- current_origin = f"slow_update_epoch_{ epoch :02d} "
1640-
1641- print (
1642- f" [slow update] force-injected into current & best "
1643- f"({ len (slow_content )} chars), "
1644- f"{ slow_time } s"
1645- )
1640+ # Slow update acceptance — two modes selected via
1641+ # `optimizer.slow_update_gate_with_selection`.
1642+ if slow_gate_with_selection :
1643+ # ── Gated mode (follow SkillReflection) ──────────
1644+ # Evaluate the slow-update candidate on the
1645+ # selection set and accept/reject via the same
1646+ # validation gate used for step-level updates.
1647+ if slow_candidate_hash in sel_cache :
1648+ slow_sel_hard , slow_sel_soft = sel_cache [
1649+ slow_candidate_hash
1650+ ]
1651+ print (
1652+ f" [slow gate] cache hit: "
1653+ f"hard={ slow_sel_hard :.4f} "
1654+ )
1655+ else :
1656+ sel_env , sel_n = _build_eval_env (
1657+ split = "valid_seen" ,
1658+ env_num = cfg ["sel_env_num" ],
1659+ seed = seed ,
1660+ )
1661+ print (f" [slow gate] selection items={ sel_n } " )
1662+ slow_eval_dir = os .path .join (
1663+ slow_dir , "selection_eval" ,
1664+ )
1665+ slow_eval_results = adapter .rollout (
1666+ sel_env , slow_candidate , slow_eval_dir ,
1667+ )
1668+ slow_sel_hard , slow_sel_soft = compute_score (
1669+ slow_eval_results
1670+ )
1671+ sel_cache [slow_candidate_hash ] = (
1672+ slow_sel_hard , slow_sel_soft ,
1673+ )
1674+
1675+ slow_gate = evaluate_gate (
1676+ candidate_skill = slow_candidate ,
1677+ cand_hard = slow_sel_hard ,
1678+ current_skill = current_skill ,
1679+ current_score = current_score ,
1680+ best_skill = best_skill ,
1681+ best_score = best_score ,
1682+ best_step = best_step ,
1683+ global_step = global_step ,
1684+ cand_soft = slow_sel_soft ,
1685+ metric = gate_metric ,
1686+ mixed_weight = gate_mixed_weight ,
1687+ )
1688+ slow_result ["selection_hard" ] = slow_sel_hard
1689+ slow_result ["selection_soft" ] = slow_sel_soft
1690+ slow_result ["action" ] = slow_gate .action
1691+ prev_current = current_score
1692+ prev_best = best_score
1693+ current_skill = slow_gate .current_skill
1694+ current_score = slow_gate .current_score
1695+ best_skill = slow_gate .best_skill
1696+ best_score = slow_gate .best_score
1697+ best_step = slow_gate .best_step
1698+ if slow_gate .action in {"accept" , "accept_new_best" }:
1699+ current_origin = (
1700+ f"slow_update_epoch_{ epoch :02d} "
1701+ )
1702+ if slow_gate .action == "accept_new_best" :
1703+ best_origin = current_origin
1704+ print (
1705+ f" [slow gate] ACCEPT (new best) "
1706+ f"hard={ slow_sel_hard :.4f} > "
1707+ f"prev best { prev_best :.4f} "
1708+ )
1709+ elif slow_gate .action == "accept" :
1710+ print (
1711+ f" [slow gate] ACCEPT "
1712+ f"hard={ slow_sel_hard :.4f} > "
1713+ f"current={ prev_current :.4f} "
1714+ )
1715+ else :
1716+ print (
1717+ f" [slow gate] REJECT "
1718+ f"hard={ slow_sel_hard :.4f} <= "
1719+ f"current={ current_score :.4f} "
1720+ )
1721+ print (
1722+ f" [slow update] guidance written "
1723+ f"({ len (slow_result ['slow_update_content' ])} "
1724+ f"chars), { slow_time } s"
1725+ )
1726+ else :
1727+ # ── Force-accept mode (default) ──────────────────
1728+ # The epoch-level longitudinal guidance is injected
1729+ # into both current_skill and best_skill
1730+ # unconditionally — it must not be gated by
1731+ # step-level selection scores.
1732+ slow_content = slow_result ["slow_update_content" ]
1733+ current_skill = replace_slow_update_field (
1734+ current_skill , slow_content ,
1735+ )
1736+ best_skill = replace_slow_update_field (
1737+ best_skill , slow_content ,
1738+ )
1739+ # Update caches so downstream steps use the
1740+ # slow-update-injected skill for hashing.
1741+ slow_candidate_hash = skill_hash (current_skill )
1742+ sel_cache [slow_candidate_hash ] = (current_score , 0.0 )
1743+
1744+ slow_result ["action" ] = "force_accept"
1745+ current_origin = f"slow_update_epoch_{ epoch :02d} "
1746+
1747+ print (
1748+ f" [slow update] force-injected into "
1749+ f"current & best "
1750+ f"({ len (slow_content )} chars), "
1751+ f"{ slow_time } s"
1752+ )
16461753 else :
16471754 slow_result = slow_result or {}
16481755 slow_result ["action" ] = "no_content"
0 commit comments