diff --git a/README.md b/README.md
index 050c64d9..60534512 100644
--- a/README.md
+++ b/README.md
@@ -15,6 +15,108 @@
[](https://itmo.ru/)
[](https://t.me/+0kMcymeAQrczN2Fi)
+---
+
+
+
+# 𧬠CoEvo β ΡΡΡΡΠΊΡΡΡΠ½Π°Ρ ΠΊΠΎ-ΡΠ²ΠΎΠ»ΡΡΠΈΡ ΠΏΡΠΎΠΌΠΏΡΠΎΠ²
+
+**ΠΡΠΏΡΡΠΊΠ½Π°Ρ ΠΊΠ²Π°Π»ΠΈΡΠΈΠΊΠ°ΡΠΈΠΎΠ½Π½Π°Ρ ΡΠ°Π±ΠΎΡΠ°**
+
+ΠΠ΅ΡΠΎΠ΄ Π°Π²ΡΠΎΠΌΠ°ΡΠΈΡΠ΅ΡΠΊΠΎΠΉ ΠΎΠΏΡΠΈΠΌΠΈΠ·Π°ΡΠΈΠΈ ΠΏΡΠΎΠΌΠΏΡΠΎΠ², ΡΠ΅Π°Π»ΠΈΠ·ΠΎΠ²Π°Π½Π½ΡΠΉ Π²Π½ΡΡΡΠΈ ΡΡΠ΅ΠΉΠΌΠ²ΠΎΡΠΊΠ° CoolPrompt
+
+
+
+> ΠΡΠΎΡ ΡΠ°Π·Π΄Π΅Π» (Π²Π΅ΡΠΊΠ° `role_based`) ΠΎΠΏΠΈΡΡΠ²Π°Π΅Ρ ΠΌΠΎΠΉ Π΄ΠΈΠΏΠ»ΠΎΠΌΠ½ΡΠΉ ΠΌΠ΅ΡΠΎΠ΄ **CoEvo** ΠΈ Π΅Π³ΠΎ ΡΡΠΈΠ»Π΅Π½Π½ΡΡ Π²Π΅ΡΡΠΈΡ **CoEvo-M**: ΠΈΠ΄Π΅Ρ, ΡΠ΅Π·ΡΠ»ΡΡΠ°ΡΡ, Π·Π°ΠΏΡΡΠΊΠ°Π΅ΠΌΠΎΠ΅ Π΄Π΅ΠΌΠΎ ΠΈ ΡΠΏΠΈΡΠΎΠΊ ΠΊΠ»ΡΡΠ΅Π²ΡΡ
ΡΠ°ΠΉΠ»ΠΎΠ². ΠΠ±ΡΠ΅Π΅ ΠΎΠΏΠΈΡΠ°Π½ΠΈΠ΅ ΡΡΠ΅ΠΉΠΌΠ²ΠΎΡΠΊΠ° CoolPrompt β [Π½ΠΈΠΆΠ΅](#coolprompt-framework).
+
+## Π ΡΡΠΌ ΠΈΠ΄Π΅Ρ
+
+ΠΠ»Π°ΡΡΠΈΡΠ΅ΡΠΊΠΈΠ΅ ΠΌΠ΅ΡΠΎΠ΄Ρ ΠΎΠΏΡΠΈΠΌΠΈΠ·ΠΈΡΡΡΡ ΠΏΡΠΎΠΌΠΏΡ ΠΊΠ°ΠΊ **Π΅Π΄ΠΈΠ½ΡΠΉ ΠΊΡΡΠΎΠΊ ΡΠ΅ΠΊΡΡΠ°**. CoEvo ΠΏΡΠ΅Π΄ΡΡΠ°Π²Π»ΡΠ΅Ρ ΠΏΡΠΎΠΌΠΏΡ ΠΊΠ°ΠΊ **ΡΡΠΈ Π½Π΅Π·Π°Π²ΠΈΡΠΈΠΌΡΡ
ΠΏΠΎΠ»Ρ** ΠΈ ΡΠ²ΠΎΠ»ΡΡΠΈΠΎΠ½ΠΈΡΡΠ΅Ρ ΠΊΠ°ΠΆΠ΄ΠΎΠ΅ ΠΈΠ· Π½ΠΈΡ
:
+
+| ΠΠΎΠ»Π΅ | ΠΡΠ΄Π° ΠΏΠΎΠ΄ΡΡΠ°Π²Π»ΡΠ΅ΡΡΡ | ΠΠ° ΡΡΠΎ ΠΎΡΠ²Π΅ΡΠ°Π΅Ρ |
+|------|--------------------|-----------------|
+| `role` (`system_behavior`) | **system**-ΡΠΎΠΎΠ±ΡΠ΅Π½ΠΈΠ΅ | ΡΠΎΠ»Ρ ΠΈ ΠΏΠΎΠ²Π΅Π΄Π΅Π½ΠΈΠ΅ ΠΌΠΎΠ΄Π΅Π»ΠΈ |
+| `task` (`task_description`) | **user**-ΡΠΎΠΎΠ±ΡΠ΅Π½ΠΈΠ΅ | ΡΡΠΎ ΠΈΠΌΠ΅Π½Π½ΠΎ Π½ΡΠΆΠ½ΠΎ ΡΠ΄Π΅Π»Π°ΡΡ |
+| `constraints` (`output_constraints`) | **user**-ΡΠΎΠΎΠ±ΡΠ΅Π½ΠΈΠ΅ | ΡΠΎΡΠΌΠ°Ρ ΠΈ ΠΎΠ³ΡΠ°Π½ΠΈΡΠ΅Π½ΠΈΡ ΠΎΡΠ²Π΅ΡΠ° |
+
+1. **ΠΠ΅ΠΊΠΎΠΌΠΏΠΎΠ·ΠΈΡΠΈΡ.** ΠΠ° ΡΡΠ°ΡΡΠ΅ ΠΎΠ΄ΠΈΠ½ Π²ΡΠ·ΠΎΠ² LLM-ΠΎΠΏΡΠΈΠΌΠΈΠ·Π°ΡΠΎΡΠ° ΡΠ°ΡΠΊΠ»Π°Π΄ΡΠ²Π°Π΅Ρ ΠΈΡΡ
ΠΎΠ΄Π½ΡΠΉ ΠΏΡΠΎΠΌΠΏΡ Π½Π° ΡΡΠΎΠΉΠΊΡ `role / task / constraints` (ΡΡΡΡΠΊΡΡΡΠΈΡΠΎΠ²Π°Π½Π½ΡΠΉ JSON).
+2. **ΠΠ²ΠΎΠ»ΡΡΠΈΡ.** ΠΠΎΠΏΡΠ»ΡΡΠΈΡ ΡΠ°ΠΊΠΈΡ
ΡΡΠΎΠ΅ΠΊ ΠΎΠΏΡΠΈΠΌΠΈΠ·ΠΈΡΡΠ΅ΡΡΡ Π³Π΅Π½Π΅ΡΠΈΡΠ΅ΡΠΊΠΈ: ΡΡΠ»Π΅ΡΠΎΡΠ½ΡΠΉ ΠΎΡΠ±ΠΎΡ β ΡΠ΅ΡΠ»Π΅ΠΊΡΠΈΡ β ΠΊΡΠΎΡΡΠΎΠ²Π΅Ρ β ΠΌΡΡΠ°ΡΠΈΡ β softmax-Π²ΡΠΆΠΈΠ²Π°Π½ΠΈΠ΅. Π Π΅ΡΠ»Π΅ΠΊΡΠΈΡ ΠΎΠ±ΡΡΡΠ½ΡΠ΅Ρ ΠΌΠΎΠ΄Π΅Π»ΠΈ, *ΡΠ΅ΠΌ* ΡΠ΄Π°ΡΠ½ΡΠ΅ Π²Π°ΡΠΈΠ°Π½ΡΡ Π»ΡΡΡΠ΅ Π½Π΅ΡΠ΄Π°ΡΠ½ΡΡ
.
+3. **ΠΡΠ±ΠΎΡ ΠΏΠΎΠ»Π΅ΠΉ.** Π ΠΊΠΎΠ½ΡΠ΅ ablation Π½Π° Π²Π°Π»ΠΈΠ΄Π°ΡΠΈΠΈ Π²ΡΠ±ΠΈΡΠ°Π΅Ρ Π»ΡΡΡΡΡ ΠΊΠΎΠΌΠ±ΠΈΠ½Π°ΡΠΈΡ ΠΏΠΎΠ»Π΅ΠΉ (task / task+role / task+role+constraints).
+
+**CoEvo-M** β ΡΡΠΈΠ»Π΅Π½Π½Π°Ρ Π²Π΅ΡΡΠΈΡ: ΡΡΡΠ°Ρ Π·Π° Π΄Π»ΠΈΠ½Ρ ΡΠΎΠ»ΠΈ ΠΈ Π·Π° ΡΠΌΡΡΠ»ΠΎΠ²ΠΎΠ΅ Π΄ΡΠ±Π»ΠΈΡΠΎΠ²Π°Π½ΠΈΠ΅ `role`/`task` (sentence-transformers), hall-of-fame Π»ΡΡΡΠΈΡ
ΠΎΡΠΎΠ±Π΅ΠΉ, Β«ΠΏΠ»ΠΎΡ
ΠΈΠ΅ ΠΏΡΠΈΠΌΠ΅ΡΡΒ» Π² ΠΌΡΡΠ°ΡΠΈΠΈ ΠΈ ΡΠΎΡΡΠΈΡΠΎΠ²Π°Π½Π½ΡΠΉ ΡΠ»ΠΈΡΠΈΠ·ΠΌ.
+
+## Π Π΅Π·ΡΠ»ΡΡΠ°ΡΡ
+
+Π‘ΡΠ°Π²Π½Π΅Π½ΠΈΠ΅ Ρ Π±Π°Π·ΠΎΠ²ΡΠΌ ReflectivePrompt Π½Π° 6 Π΄Π°ΡΠ°ΡΠ΅ΡΠ°Ρ
(BERTScore / ΠΌΠ΅ΡΡΠΈΠΊΠ° Π·Π°Π΄Π°ΡΠΈ):
+
+
+
+
+
+| ΠΠ°ΡΠ°ΡΠ΅Ρ | ReflectivePrompt | CoEvo | CoEvo-M |
+|---------|:---:|:---:|:---:|
+| TweetEval | 0.705 | **0.726** | 0.719 |
+| SQuAD v2 | 0.878 | 0.907 | **0.929** |
+| CommonGen | 0.808 | **0.809** | 0.807 |
+| MEDIQA | 0.688 | 0.700 | **0.703** |
+| GSM8K | 0.919 | **0.927** | 0.926 |
+| XSum | 0.730 | **0.736** | 0.734 |
+| **Π‘ΡΠ΅Π΄Π½Π΅Π΅** | 0.788 | 0.801 | **0.803** |
+
+## ΠΠ°ΠΏΡΡΠΊΠ°Π΅ΠΌΠΎΠ΅ Π΄Π΅ΠΌΠΎ
+
+π **[notebooks/examples/coevo_demo.ipynb](notebooks/examples/coevo_demo.ipynb)** β CoEvo end-to-end ΡΠ»ΡΡΡΠ°Π΅Ρ ΠΏΡΠΎΠΌΠΏΡ Π΄Π»Ρ QA ΠΏΠΎ SQuAD v2.
+
+ΠΠ΄Π΅Ρ ΡΡΠ΅Π½Π°ΡΠΈΡ: ΡΠΈΠ»ΡΠ½ΡΠΉ ΠΎΠΏΡΠΈΠΌΠΈΠ·Π°ΡΠΎΡ (`gpt-4o-mini`) ΠΏΠ΅ΡΠ΅ΠΏΠΈΡΡΠ²Π°Π΅Ρ ΠΏΡΠΎΠΌΠΏΡ Π΄Π»Ρ Π΄Π΅ΡΡΠ²ΠΎΠΉ ΠΏΡΠΎΠ΄Π°ΠΊΡΠ½-ΠΌΠΎΠ΄Π΅Π»ΠΈ (`gpt-4.1-nano`).
+
+
+
+
+
+ΠΠ· ΠΏΡΠΎΡΡΠΎΠ³ΠΎ `"Answer the question based on the context."` ΠΌΠ΅ΡΠΎΠ΄ Π·Π° 5 ΡΠΏΠΎΡ
ΡΠΎΠ±ΠΈΡΠ°Π΅Ρ ΡΡΡΡΠΊΡΡΡΠΈΡΠΎΠ²Π°Π½Π½ΡΠΉ ΠΏΡΠΎΠΌΠΏΡ ΠΈ ΠΏΠΎΠ΄Π½ΠΈΠΌΠ°Π΅Ρ BERTScore **0.823 β 0.896 (+0.073)**.
+
+## ΠΡΡΡΡΡΠΉ ΡΡΠ°ΡΡ CoEvo
+
+```python
+from coolprompt.assistant import PromptTuner
+
+tuner = PromptTuner() # OPENAI_API_KEY Π² ΠΎΠΊΡΡΠΆΠ΅Π½ΠΈΠΈ
+
+tuner.run(
+ start_prompt="Answer the question based on the context.",
+ task="generation",
+ metric="bertscore",
+ dataset=dataset, # ΡΠΏΠΈΡΠΎΠΊ Π²Ρ
ΠΎΠ΄ΠΎΠ²
+ target=target, # ΡΠΏΠΈΡΠΎΠΊ ΡΡΠ°Π»ΠΎΠ½Π½ΡΡ
ΠΎΡΠ²Π΅ΡΠΎΠ²
+ method="coevo", # CoEvo-M ΠΏΠΎ ΡΠΌΠΎΠ»ΡΠ°Π½ΠΈΡ; use_enhancements=False Π±Π°Π·ΠΎΠ²ΡΠΉ CoEvo
+)
+
+# CoEvo Π²ΠΎΠ·Π²ΡΠ°ΡΠ°Π΅Ρ ΡΡΠΈ ΠΏΠΎΠ»Ρ:
+print(tuner.final_role) # ΡΠΎΠ»Ρ
+print(tuner.final_prompt) # Π·Π°Π΄Π°ΡΠ°
+print(tuner.final_constraints) # ΠΎΠ³ΡΠ°Π½ΠΈΡΠ΅Π½ΠΈΡ ΡΠΎΡΠΌΠ°ΡΠ°
+```
+
+## ΠΠΎΠΉ Π²ΠΊΠ»Π°Π΄ ΠΈ ΠΊΠ»ΡΡΠ΅Π²ΡΠ΅ ΡΠ°ΠΉΠ»Ρ
+
+Π Π΅Π°Π»ΠΈΠ·Π°ΡΠΈΡ ΠΌΠ΅ΡΠΎΠ΄ΠΎΠ² CoEvo / CoEvo-M ΠΏΠΎΠ²Π΅ΡΡ
ΡΡΠ΅ΠΉΠΌΠ²ΠΎΡΠΊΠ° CoolPrompt:
+
+- **[`coolprompt/optimizer/reflective_prompt/coevo_base_evoluter.py`](coolprompt/optimizer/reflective_prompt/coevo_base_evoluter.py)** β ΡΠ΄ΡΠΎ ΠΌΠ΅ΡΠΎΠ΄Π°: Π΄Π΅ΠΊΠΎΠΌΠΏΠΎΠ·ΠΈΡΠΈΡ ΠΏΡΠΎΠΌΠΏΡΠ° Π½Π° 3 ΠΏΠΎΠ»Ρ, ΡΠ²ΠΎΠ»ΡΡΠΈΠΎΠ½Π½ΡΠΉ ΡΠΈΠΊΠ», ΡΠ΅ΡΠ»Π΅ΠΊΡΠΈΡ, ΠΎΡΠ±ΠΎΡ ΠΏΠΎΠ»Π΅ΠΉ.
+- **[`coolprompt/optimizer/reflective_prompt/coevo_evoluter.py`](coolprompt/optimizer/reflective_prompt/coevo_evoluter.py)** β ΠΎΠΏΠ΅ΡΠ°ΡΠΎΡΡ ΠΊΡΠΎΡΡΠΎΠ²Π΅ΡΠ° ΠΈ ΠΌΡΡΠ°ΡΠΈΠΈ Π½Π° ΡΡΠΎΠ²Π½Π΅ ΠΏΠΎΠ»Π΅ΠΉ.
+- **[`coolprompt/optimizer/reflective_prompt/factorized_evoluter.py`](coolprompt/optimizer/reflective_prompt/factorized_evoluter.py)** β ΡΠ°ΠΊΡΠΎΡΠΈΠ·ΠΎΠ²Π°Π½Π½Π°Ρ ΡΠ²ΠΎΠ»ΡΡΠΈΡ ΠΏΠΎ ΠΎΡΠ΄Π΅Π»ΡΠ½ΡΠΌ ΠΏΠΎΠ»ΡΠΌ.
+- **[`coolprompt/optimizer/reflective_prompt/run.py`](coolprompt/optimizer/reflective_prompt/run.py)** β `CoevoMethod`, ΠΈΠ½ΡΠ΅Π³ΡΠ°ΡΠΈΡ Π² ΠΏΡΠ±Π»ΠΈΡΠ½ΡΠΉ API (`method="coevo"`).
+- **[`coolprompt/utils/prompt_templates/`](coolprompt/utils/prompt_templates/)** β ΠΌΠ΅ΡΠ°-ΠΏΡΠΎΠΌΠΏΡΡ CoEvo: `reflective_templates_coevo_enhanced.py`, `reflective_templates_coevo_per_field.py`, `reflective_templates_coevolution.py`.
+
+
+## ΠΠ°ΡΠ΅ΡΠΈΠ°Π»Ρ
+- π ΠΠ΅ΠΌΠΎ-Π½ΠΎΡΡΠ±ΡΠΊ: [coevo_demo.ipynb](notebooks/examples/coevo_demo.ipynb).
+
+---
+
+
+
+# CoolPrompt β ΡΡΠ΅ΠΉΠΌΠ²ΠΎΡΠΊ Π°Π²ΡΠΎΠΏΡΠΎΠΌΠΏΡΠΈΠ½Π³Π°
+
CoolPrompt is a framework for automatic prompt creation and optimization.
### Join our [telegram](https://t.me/+0kMcymeAQrczN2Fi) channel to be in touch.
diff --git a/coolprompt/assistant.py b/coolprompt/assistant.py
index 6c86b101..9345fde0 100644
--- a/coolprompt/assistant.py
+++ b/coolprompt/assistant.py
@@ -61,6 +61,8 @@ def __init__(
self.init_prompt = None
self.final_metric = None
self.final_prompt = None
+ self.final_role = None
+ self.final_constraints = None
self.assistant_feedback = None
self.synthetic_dataset = None
@@ -323,6 +325,11 @@ def run(
**kwargs,
)
+ self.final_role = getattr(method_impl, "last_role", "") or None
+ self.final_constraints = (
+ getattr(method_impl, "last_constraints", "") or None
+ )
+
logger.info("Running the prompt format checking...")
final_prompt = correct(
prompt=final_prompt,
@@ -346,6 +353,8 @@ def run(
dataset=dataset_split[1],
targets=dataset_split[3],
template=template,
+ system_role=self.final_role,
+ constraints=self.final_constraints,
)
logger.info(
f"Initial {base_metric} score: {self.init_metric}, "
diff --git a/coolprompt/evaluator/evaluator.py b/coolprompt/evaluator/evaluator.py
index f6f059dc..283df7f5 100644
--- a/coolprompt/evaluator/evaluator.py
+++ b/coolprompt/evaluator/evaluator.py
@@ -6,6 +6,7 @@
from langchain_core.language_models.base import BaseLanguageModel
from langchain_core.messages.ai import AIMessage
+from langchain_core.messages import SystemMessage, HumanMessage
import numpy as np
from coolprompt.evaluator.metrics import BaseMetric
from coolprompt.utils.logging_config import logger
@@ -67,6 +68,8 @@ def evaluate(
targets: list[str | int],
template: Optional[str] = None,
failed_examples: Optional[int] = None,
+ system_role: Optional[str] = None,
+ constraints: Optional[str] = None,
*,
return_detailed: bool = False,
save_model_answers: bool = False,
@@ -112,7 +115,9 @@ def evaluate(
if self.task == Task.CLASSIFICATION:
self.metric.extract_labels(targets)
full_prompts = [
- self._get_full_prompt(prompt, sample, template)
+ self._get_full_prompt(
+ prompt, sample, template, system_role, constraints
+ )
for sample in dataset
]
@@ -203,7 +208,9 @@ def _get_full_prompt(
prompt: str,
sample: str,
template: Optional[str] = None,
- ) -> str:
+ system_role: Optional[str] = None,
+ constraints: Optional[str] = None,
+ ) -> str | list:
"""Inserts parts of the prompt into the task template.
Args:
@@ -212,25 +219,43 @@ def _get_full_prompt(
template (Optional[str]):
Prompt template for defined task type.
If None, uses default template.
+ system_role (Optional[str]): system behavior prepended as a
+ SystemMessage (CoEvo). Defaults to None.
+ constraints (Optional[str]): output format constraints appended
+ to the prompt (CoEvo). Defaults to None.
Raises:
ValueError: if type of task is not supported
Returns:
- str: the full prompt to be passed to the model
+ str | list: the full prompt string, or a list of
+ SystemMessage + HumanMessage if system_role is set.
"""
if template is None:
template = self._get_default_template()
+ effective_prompt = prompt
+ if constraints:
+ effective_prompt = f"{prompt}\n\n{constraints}"
+
match self.task:
case Task.CLASSIFICATION:
labels = ", ".join(map(str, self.metric.label_to_id.keys()))
- return template.format(
- PROMPT=prompt, LABELS=labels, INPUT=sample
+ formatted = template.format(
+ PROMPT=effective_prompt, LABELS=labels, INPUT=sample
)
case Task.GENERATION:
- return template.format(PROMPT=prompt, INPUT=sample)
+ formatted = template.format(
+ PROMPT=effective_prompt, INPUT=sample
+ )
+
+ if system_role:
+ return [
+ SystemMessage(content=system_role),
+ HumanMessage(content=formatted),
+ ]
+ return formatted
def _get_default_template(self) -> str:
"""Returns the default template for the task type."""
diff --git a/coolprompt/evaluator/metrics.py b/coolprompt/evaluator/metrics.py
index e1271341..d39e79b9 100644
--- a/coolprompt/evaluator/metrics.py
+++ b/coolprompt/evaluator/metrics.py
@@ -73,6 +73,10 @@ def _compute_raw(
List[float]: List of float metrics (for each model answer).
"""
+ outputs = [
+ "none" if isinstance(o, str) and not o.strip() else o
+ for o in outputs
+ ]
return [
self._postprocessing(
self._metric.compute(
diff --git a/coolprompt/optimizer/reflective_prompt/__init__.py b/coolprompt/optimizer/reflective_prompt/__init__.py
index fd0b6dcc..31770bd8 100644
--- a/coolprompt/optimizer/reflective_prompt/__init__.py
+++ b/coolprompt/optimizer/reflective_prompt/__init__.py
@@ -1,3 +1,19 @@
-from coolprompt.optimizer.reflective_prompt.run import ReflectiveMethod, reflectiveprompt
-
-__all__ = ["reflectiveprompt", "ReflectiveMethod"]
+from coolprompt.optimizer.reflective_prompt.run import (
+ ReflectiveMethod,
+ reflectiveprompt,
+ coevo,
+ CoevoMethod,
+)
+from coolprompt.optimizer.reflective_prompt.factorized_evoluter import (
+ FactorizedEvoluter,
+)
+from coolprompt.optimizer.reflective_prompt.coevo_evoluter import CoevoEvoluter
+
+__all__ = [
+ "reflectiveprompt",
+ "ReflectiveMethod",
+ "coevo",
+ "CoevoMethod",
+ "FactorizedEvoluter",
+ "CoevoEvoluter",
+]
diff --git a/coolprompt/optimizer/reflective_prompt/coevo_base_evoluter.py b/coolprompt/optimizer/reflective_prompt/coevo_base_evoluter.py
new file mode 100644
index 00000000..be3e055a
--- /dev/null
+++ b/coolprompt/optimizer/reflective_prompt/coevo_base_evoluter.py
@@ -0,0 +1,1268 @@
+import os
+import time
+import yaml
+from typing import Dict, List, Optional, Tuple, Any
+
+import numpy as np
+import statistics
+from scipy.special import softmax
+from sklearn.metrics.pairwise import cosine_similarity
+
+from langchain_core.messages.ai import AIMessage
+from langchain_core.language_models.base import BaseLanguageModel
+
+from coolprompt.evaluator import Evaluator
+from coolprompt.optimizer.reflective_prompt.prompt import Prompt, PromptOrigin
+from coolprompt.utils.logging_config import logger
+
+from coolprompt.utils.prompt_templates.reflective_templates_fixed_role import (
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_FIXED_ROLE,
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_FIXED_ROLE,
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_FIXED_ROLE,
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_FIXED_ROLE,
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_FIXED_ROLE,
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_FIXED_ROLE,
+)
+from coolprompt.utils.prompt_templates.reflective_templates_no_role import (
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_NO_ROLE,
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_NO_ROLE,
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_NO_ROLE,
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_NO_ROLE,
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_NO_ROLE,
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_NO_ROLE,
+)
+from coolprompt.utils.prompt_templates.reflective_templates_coevolution import (
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO,
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO,
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO,
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO,
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO,
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO,
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_BASE,
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_BASE,
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_BASE,
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_3F,
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO_3F,
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_3F,
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_3F,
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO_3F,
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_3F,
+)
+from coolprompt.utils.prompt_templates.reflective_templates_text_only import (
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_TEXT_ONLY,
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_TEXT_ONLY,
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_TEXT_ONLY,
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_TEXT_ONLY,
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_TEXT_ONLY,
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_TEXT_ONLY,
+)
+from coolprompt.utils.prompt_templates.reflective_templates_factorized import (
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_ROLE_ONLY,
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_ROLE_ONLY,
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_ROLE_ONLY,
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_ROLE_ONLY,
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_ROLE_ONLY,
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_CONSTRAINTS_ONLY,
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY,
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY,
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_CONSTRAINTS_ONLY,
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_CONSTRAINTS_ONLY,
+)
+from coolprompt.utils.parsing import extract_answer, extract_json
+
+_embedding_model = None
+
+
+def _get_embedding_model():
+ global _embedding_model
+ if _embedding_model is None:
+ from sentence_transformers import SentenceTransformer
+
+ _embedding_model = SentenceTransformer("all-MiniLM-L6-v2")
+ return _embedding_model
+
+
+class ReflectiveEvoluter:
+ """
+ ReflectiveEvoluter class that represents evoluter for ReflectivePrompt
+
+ Attributes:
+ model: langchain.BaseLanguageModel class of model to use.
+ evaluator: evaluator (Evaluator) to compute metrics.
+ train_dataset: a dataset to use while training.
+ train_targets: string targets for train dataset.
+ validation_dataset: a dataset to use while validating final prompts.
+ validation_targets: string targets for validation dataset.
+ problem_description: a string that contains
+ short description of problem to optimize.
+ initial_prompt: initial prompt to start evolution from.
+ Will be automatically generated if not provided.
+ Defaults to None.
+ population_size: an integer fixed size of prompt population.
+ Defaults to 10.
+ num_epochs: an integer number of epochs to evaluate.
+ Defaults to 10.
+ use_cache: a boolean variable.
+ Either to use caching files or not.
+ output_path: a path to store logs of evolution.
+ elitist: a prompt with highest score in population.
+ best_score_overall: best evaluation score during evolution.
+ best_prompt_overall: text of prompt with best score overall.
+ iteration: current iteration (epoch) of evolution.
+ PROMPT_TAGS: start and end tags for prompt extraction.
+ HINT_TAGS: start and end tags for hint extraction.
+ """
+
+ PROMPT_TAGS = ("", "")
+ HINT_TAGS = ("", "")
+ ROLE_LENGTH_ALPHA: float = 0.02
+ ROLE_PROMPT_SIM_THRESHOLD: float = 0.72
+ ROLE_PROMPT_SIM_ALPHA: float = 0.05
+ ELITIST_MAX_FREEZE: int = 3
+ BAD_EXAMPLES_TOP_K: int = 3
+ PREVIEW_LEN: int = 80
+ HALL_OF_FAME_MIN_SIZE: int = 10
+
+ def __init__(
+ self,
+ model: BaseLanguageModel,
+ evaluator: Evaluator,
+ train_dataset: List[str],
+ train_targets: List[str],
+ validation_dataset: List[str],
+ validation_targets: List[str],
+ problem_description: str,
+ initial_prompt: Optional[str] = None,
+ initial_role: Optional[str] = None,
+ initial_constraints: Optional[str] = None,
+ evolve_role: bool = True,
+ evolve_constraints: bool = False,
+ population_size: int = 10,
+ num_epochs: int = 10,
+ output_path: str = "./reflectiveprompt_outputs",
+ use_cache: bool = True,
+ use_enhancements: bool = True,
+ use_bad_examples: Optional[bool] = None,
+ freeze_text: bool = False,
+ text_only: bool = False,
+ val_evaluator: Optional[Evaluator] = None,
+ ) -> None:
+ self.model = model
+ self.evaluator = evaluator
+ self.val_evaluator = val_evaluator or evaluator
+ self.train_dataset = train_dataset
+ self.train_targets = train_targets
+ self.validation_dataset = validation_dataset
+ self.validation_targets = validation_targets
+ self.use_cache = use_cache
+ self.population_size = population_size
+ self.num_epochs = num_epochs
+ self.problem_description = problem_description
+ self.output_path = output_path
+ self.initial_prompt = initial_prompt
+ self.initial_role = initial_role
+ self.initial_constraints = initial_constraints or ""
+ self.evolve_role = evolve_role
+ self.evolve_constraints = evolve_constraints
+ self.use_enhancements = use_enhancements
+ self.use_bad_examples = (
+ use_enhancements if use_bad_examples is None else use_bad_examples
+ )
+ self.freeze_text = freeze_text
+ self.text_only = text_only
+ self._role_only = (
+ self.evolve_role
+ and self.freeze_text
+ and not self.evolve_constraints
+ )
+ self._constraints_only = (
+ not self.evolve_role
+ and bool(self.initial_role)
+ and self.evolve_constraints
+ )
+
+ self.elitist = None
+ self._long_term_reflection_str = ""
+ self.best_score_overall = None
+ self.best_prompt_overall = None
+ self.best_role_overall = None
+ self.best_constraints_overall = None
+ self.iteration = 0
+ self._elitist_freeze_count: int = 0
+ self._prev_elitist_role: str = ""
+ self._hall_of_fame: List[Prompt] = []
+ self._elitist_bad_examples: List[Dict] = []
+
+ self._setup_templates()
+
+ def _setup_templates(self) -> None:
+ """Selects prompt templates based on the active evolution mode."""
+ if self.text_only:
+ self._paraphrasing_template = (
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_TEXT_ONLY
+ )
+ self._crossover_template = (
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_TEXT_ONLY
+ )
+ self._mutation_template = (
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_TEXT_ONLY
+ )
+ self._short_term_template = (
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_TEXT_ONLY
+ )
+ self._long_term_template = (
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_TEXT_ONLY
+ )
+ self._initial_prompt_template = (
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_TEXT_ONLY
+ )
+ elif self._role_only:
+ self._paraphrasing_template = (
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_ROLE_ONLY
+ )
+ self._crossover_template = (
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_ROLE_ONLY
+ )
+ self._mutation_template = (
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_ROLE_ONLY
+ )
+ self._short_term_template = (
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_ROLE_ONLY
+ )
+ self._long_term_template = (
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_ROLE_ONLY
+ )
+ self._initial_prompt_template = (
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO
+ )
+ elif self._constraints_only:
+ self._paraphrasing_template = (
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_CONSTRAINTS_ONLY
+ )
+ self._crossover_template = (
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_CONSTRAINTS_ONLY
+ )
+ self._mutation_template = (
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_CONSTRAINTS_ONLY
+ )
+ self._short_term_template = (
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY
+ )
+ self._long_term_template = (
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY
+ )
+ self._initial_prompt_template = (
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO
+ )
+ elif self.evolve_role and self.evolve_constraints:
+ self._paraphrasing_template = (
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO_3F
+ )
+ self._crossover_template = (
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_3F
+ )
+ self._mutation_template = (
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_3F
+ )
+ self._short_term_template = (
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_3F
+ )
+ self._long_term_template = (
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO_3F
+ )
+ self._initial_prompt_template = (
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_3F
+ )
+ elif self.evolve_role:
+ self._paraphrasing_template = (
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO
+ )
+ if self.use_enhancements:
+ self._crossover_template = (
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO
+ )
+ self._mutation_template = (
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO
+ )
+ self._short_term_template = (
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO
+ )
+ else:
+ self._crossover_template = (
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_BASE
+ )
+ self._mutation_template = (
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_BASE
+ )
+ self._short_term_template = (
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_BASE
+ )
+ self._long_term_template = (
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO
+ )
+ self._initial_prompt_template = (
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO
+ )
+ elif not self.initial_role:
+ self._paraphrasing_template = (
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_NO_ROLE
+ )
+ self._crossover_template = (
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_NO_ROLE
+ )
+ self._mutation_template = REFLECTIVEPROMPT_MUTATION_TEMPLATE_NO_ROLE
+ self._short_term_template = (
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_NO_ROLE
+ )
+ self._long_term_template = (
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_NO_ROLE
+ )
+ self._initial_prompt_template = (
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_NO_ROLE
+ )
+ else:
+ self._paraphrasing_template = (
+ REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_FIXED_ROLE
+ )
+ self._crossover_template = (
+ REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_FIXED_ROLE
+ )
+ self._mutation_template = (
+ REFLECTIVEPROMPT_MUTATION_TEMPLATE_FIXED_ROLE
+ )
+ self._short_term_template = (
+ REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_FIXED_ROLE
+ )
+ self._long_term_template = (
+ REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_FIXED_ROLE
+ )
+ self._initial_prompt_template = (
+ REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_FIXED_ROLE
+ )
+
+ def _reranking(self, population: List[Prompt]) -> List[Prompt]:
+ """
+ Sorts given population of prompts by their scores in descending order.
+
+ Args:
+ population (List[Prompt]): population to sort.
+
+ Returns:
+ List[Prompt]: sorted population.
+ """
+ return list(
+ sorted(population, key=lambda prompt: prompt.score, reverse=True)
+ )
+
+ @staticmethod
+ def _role_prompt_sim(role: str, prompt_text: str) -> float:
+ """Cosine similarity between system_behavior and task_description.
+ Computed from all-MiniLM-L6-v2 sentence-transformer embeddings.
+ High similarity means the two components are redundant.
+ Returns 0.0 if either string is empty.
+ """
+ if not role or not prompt_text:
+ return 0.0
+ model = _get_embedding_model()
+ embs = model.encode([role, prompt_text])
+ return float(
+ cosine_similarity(embs[0].reshape(1, -1), embs[1].reshape(1, -1))[
+ 0
+ ][0]
+ )
+
+ def _update_hall_of_fame(self, population: List[Prompt]) -> None:
+ seen = {(p.text, p.role, p.constraints) for p in self._hall_of_fame}
+ for p in population:
+ if p.score is None:
+ continue
+ key = (p.text, p.role, p.constraints)
+ if key not in seen:
+ self._hall_of_fame.append(
+ Prompt(
+ text=p.text,
+ role=p.role,
+ constraints=p.constraints,
+ origin=p.origin,
+ score=p.score,
+ )
+ )
+ seen.add(key)
+ self._hall_of_fame.sort(key=lambda x: x.score, reverse=True)
+ max_size = max(self.population_size * 2, self.HALL_OF_FAME_MIN_SIZE)
+ self._hall_of_fame = self._hall_of_fame[:max_size]
+
+ def _format_bad_examples(self) -> str:
+ if not self.use_bad_examples or not self._elitist_bad_examples:
+ return "(none)"
+ lines = []
+ for i, ex in enumerate(self._elitist_bad_examples, 1):
+ inp = ex.input[:120]
+ out = ex.output
+ correct = ex.correct
+ lines.append(
+ f"{i}. Input: {inp}\n Got: {out} | Expected: {correct}"
+ )
+ return "\n".join(lines)
+
+ def _format_top_prompts_history(self, top_k: int = 5) -> str:
+ if not self._hall_of_fame:
+ return "(none)"
+ entries = self._hall_of_fame[:top_k]
+ lines = []
+ for i, p in enumerate(entries, 1):
+ score_str = self._format_score(p.score)
+ if self._role_only:
+ content = p.role or "(empty)"
+ lines.append(f"{i}. [score={score_str}] {content[:120]}")
+ elif self._constraints_only:
+ content = p.constraints or "(empty)"
+ lines.append(f"{i}. [score={score_str}] {content[:120]}")
+ elif self.evolve_role and self.evolve_constraints:
+ role = (p.role or "(empty)")[: self.PREVIEW_LEN]
+ text = (p.text or "(empty)")[: self.PREVIEW_LEN]
+ constraints = (p.constraints or "(empty)")[: self.PREVIEW_LEN]
+ lines.append(
+ f"{i}. [score={score_str}]\n system_behavior: {role}\n task_description: {text}\n output_constraints: {constraints}"
+ )
+ elif self.evolve_role:
+ role = (p.role or "(empty)")[: self.PREVIEW_LEN]
+ text = (p.text or "(empty)")[: self.PREVIEW_LEN]
+ lines.append(
+ f"{i}. [score={score_str}]\n system_behavior: {role}\n task_description: {text}"
+ )
+ else:
+ content = p.text or "(empty)"
+ lines.append(f"{i}. [score={score_str}] {content[:120]}")
+ return "\n".join(lines)
+
+ def _aggregate_bad_examples(
+ self, population: List[Prompt], top_k: int = BAD_EXAMPLES_TOP_K
+ ) -> None:
+ scored = [
+ p for p in population if p.score is not None and p.bad_examples
+ ]
+ if not scored:
+ return
+ scored.sort(key=lambda p: p.score, reverse=True)
+ top_half = scored[: max(1, len(scored) // 2)]
+ counts: Dict[str, Dict] = {}
+ for p in top_half:
+ for ex in p.bad_examples:
+ key = ex.input
+ if key not in counts:
+ counts[key] = {"count": 0, "ex": ex}
+ counts[key]["count"] += 1
+ sorted_examples = sorted(
+ counts.values(), key=lambda x: x["count"], reverse=True
+ )
+ self._elitist_bad_examples = [x["ex"] for x in sorted_examples[:top_k]]
+
+ def _format_score(self, score) -> str:
+ if not self.use_enhancements or score is None:
+ return "N/A"
+ return f"{score:.4f}"
+
+ def _evaluate(self, prompt: Prompt, split="train") -> None:
+ """Evaluates given prompt on self.dataset and records the score.
+ When evolve_role=True and split=='train':
+ - A length penalty proportional to role length is subtracted,
+ discouraging bloated roles when scores are close.
+
+ Args:
+ prompt (Prompt): a prompt to evaluate.
+ split (str, optional): Which split of dataset to use.
+ Defaults to 'train'.
+ """
+ if split == "train":
+ dataset, targets = self.train_dataset, self.train_targets
+ else:
+ dataset, targets = self.validation_dataset, self.validation_targets
+
+ eval_role = prompt.role
+ ev = self.evaluator if split == "train" else self.val_evaluator
+ result = ev.evaluate(
+ prompt=prompt.text,
+ dataset=dataset,
+ targets=targets,
+ system_role=eval_role if eval_role else None,
+ constraints=prompt.constraints if self.evolve_constraints else None,
+ failed_examples=(
+ 10 if split == "train" and self.use_bad_examples else None
+ ),
+ )
+ if isinstance(result, tuple):
+ score, bad_examples = result
+ prompt.set_bad_examples(bad_examples)
+ else:
+ score = result
+
+ if self.evolve_role and split == "train":
+ if self.use_enhancements:
+ score = score - self.ROLE_LENGTH_ALPHA * len(prompt.role) / 1000
+
+ if prompt.role:
+ sim = self._role_prompt_sim(prompt.role, prompt.text)
+ if sim > self.ROLE_PROMPT_SIM_THRESHOLD:
+ score -= self.ROLE_PROMPT_SIM_ALPHA * (
+ sim - self.ROLE_PROMPT_SIM_THRESHOLD
+ )
+
+ prompt.set_score(score)
+
+ def _evaluation(
+ self, population: List[Prompt], split: str = "train"
+ ) -> None:
+ """Evaluation operation for prompts population.
+ Evaluates every prompt in population and records the results.
+
+ Args:
+ population (List[Prompt]): population of prompts to evaluate.
+ split (str, optional): Which split of dataset to use.
+ Defaults to 'train'.
+ """
+ logger.info("Evaluating population...")
+ for prompt in population:
+ self._evaluate(prompt, split=split)
+ if split == "train":
+ self._aggregate_bad_examples(population)
+
+ def _create_initial_prompt(self) -> Tuple[str, str, str]:
+ """Creates an initial prompt according to provided problem description
+
+ Returns:
+ Tuple[str, str, str]: initial prompt
+ """
+ request = self._initial_prompt_template.format(
+ PROBLEM_DESCRIPTION=self.problem_description
+ )
+ answer = self._llm_query([request])[0]
+ extracted = extract_json(answer)
+ if extracted is None:
+ extracted = {}
+
+ if self.evolve_role:
+ role = extracted.get("system_behavior", extracted.get("role", ""))
+ else:
+ role = self.initial_role or ""
+
+ prompt = extracted.get(
+ "task_description",
+ extracted.get(
+ "prompt",
+ extract_answer(
+ answer, self.PROMPT_TAGS, format_mismatch_label=""
+ ),
+ ),
+ )
+ constraints = (
+ extracted.get("output_constraints", "")
+ if self.evolve_constraints
+ else ""
+ )
+ return role, prompt, constraints
+
+ def _init_pop(self) -> List[Prompt]:
+ """Creates initial population of prompts.
+
+ Returns:
+ List[Prompt]: initial population.
+ """
+
+ logger.info("Initializing population...")
+ if self.initial_prompt is None:
+ generated_role, self.initial_prompt, generated_constraints = (
+ self._create_initial_prompt()
+ )
+ if self.evolve_role and not self.initial_role:
+ self.initial_role = generated_role
+ if self.evolve_constraints and not self.initial_constraints:
+ self.initial_constraints = generated_constraints
+
+ if self.initial_role is None:
+ self.initial_role = ""
+
+ fmt_kwargs = {
+ "ROLE": self.initial_role,
+ "PROMPT": self.initial_prompt,
+ "NUM_PROMPTS": self.population_size,
+ "PROBLEM_DESCRIPTION": self.problem_description,
+ }
+ if self.evolve_constraints:
+ fmt_kwargs["CONSTRAINTS"] = self.initial_constraints
+ request = self._paraphrasing_template.format(**fmt_kwargs)
+ answer = self._llm_query([request])[0]
+ extracted = extract_json(answer)
+ if extracted is None or "prompts" not in extracted:
+ logger.warning(
+ "Failed to extract prompts from LLM response, using fallback"
+ )
+ prompts_data = [
+ {"role": self.initial_role, "prompt": self.initial_prompt}
+ ] * self.population_size
+ else:
+ prompts_data = extracted["prompts"]
+
+ if not isinstance(prompts_data, list) or len(prompts_data) == 0:
+ logger.warning("Invalid prompts_data format, using fallback")
+ prompts_data = [
+ {"role": self.initial_role, "prompt": self.initial_prompt}
+ ] * self.population_size
+
+ initial_population = []
+ fixed_role = self.initial_role if not self.evolve_role else None
+
+ for p_data in prompts_data:
+ if isinstance(p_data, dict):
+ if self.evolve_role:
+ role = p_data.get(
+ "system_behavior",
+ p_data.get("role", self.initial_role),
+ )
+ else:
+ role = fixed_role or ""
+ text = p_data.get(
+ "task_description",
+ p_data.get("prompt", str(p_data)),
+ )
+ constraints = (
+ p_data.get("output_constraints", "")
+ if self.evolve_constraints
+ else ""
+ )
+ if self._role_only or self._constraints_only:
+ text = self.initial_prompt
+ initial_population.append(
+ Prompt(
+ text=text,
+ role=role,
+ constraints=constraints,
+ origin=PromptOrigin.APE,
+ )
+ )
+ else:
+ role = (
+ fixed_role or self.initial_role
+ if not self.evolve_role
+ else self.initial_role
+ )
+ initial_population.append(
+ Prompt(text=p_data, role=role, origin=PromptOrigin.APE)
+ )
+
+ initial_population[-1] = Prompt(
+ text=self.initial_prompt,
+ role=(
+ fixed_role or self.initial_role
+ if not self.evolve_role
+ else self.initial_role
+ ),
+ constraints=(
+ self.initial_constraints if self.evolve_constraints else ""
+ ),
+ origin=PromptOrigin.MANUAL,
+ )
+ self._evaluation(initial_population)
+ initial_population = self._reranking(initial_population)
+ return initial_population
+
+ def _cache_data(self, data: Any, savepath: os.PathLike) -> None:
+ """Writes the data to the yaml file.
+
+ Args:
+ data (Any): data to be cached.
+ savepath (os.PathLike): a path to saving file.
+ """
+ os.makedirs(os.path.dirname(savepath), exist_ok=True)
+ with open(savepath, "w") as f:
+ yaml.dump(data, f)
+
+ def _cache_population(
+ self, population: List[Prompt], savepath: os.PathLike
+ ) -> None:
+ """Caching a population of prompts to file.
+ If self.use_cache is False this function will do nothing.
+
+ Args:
+ population (List[Prompt]): prompt population.
+ savepath (os.PathLike): a path to saving file.
+ """
+ if self.use_cache is False:
+ return
+
+ best_score = population[0].score
+ average_score = statistics.mean([prompt.score for prompt in population])
+ data = {
+ "best_score": best_score,
+ "average_score": average_score,
+ "prompts": [prompt.to_dict() for prompt in population],
+ }
+ self._cache_data(data, savepath)
+
+ def _selection(self, population: List[Prompt]) -> List[Prompt]:
+ """Provides selection operation.
+ In current implementation we want to select parents
+ with different scores.
+ But when there is difficult to do so (trial number check),
+ it will just sample anyways.
+
+ Probabilities - normalized scores.
+
+ Args:
+ population (List[Prompt]): prompt population to select from.
+
+ Returns:
+ List[Prompt]: selected prompts.
+ """
+ selected_population = []
+
+ scores = np.array([prompt.score for prompt in population])
+ scores = np.clip(scores, 0, None)
+ if np.sum(scores) == 0:
+ probas = np.ones(len(scores)) / len(scores)
+ else:
+ probas = scores / np.sum(scores)
+
+ trial = 0
+ anyways = False
+ while len(selected_population) < 2 * self.population_size:
+ parents = np.random.choice(
+ population, size=2, replace=False, p=probas
+ )
+ if parents[0].score != parents[1].score or anyways:
+ selected_population.extend(parents)
+ trial += 1
+ if trial > 1000:
+ anyways = True
+
+ return selected_population
+
+ def _survive(
+ self, population: List[Prompt], temperature: float = None
+ ) -> List[Prompt]:
+ """Final selection before going into new epoch.
+ Probabilities are based on softmax function with temperature (if set).
+
+ Args:
+ population (List[Prompt]): population to select from.
+ temperature (float, optional): temperature parameter for softmax.
+ Defaults to None.
+
+ Returns:
+ List[Prompt]: selected (survived) prompts.
+ """
+ scores = np.array([prompt.score for prompt in population])
+ if temperature is not None:
+ scores /= temperature
+ probas = softmax(scores)
+ return np.random.choice(
+ population, size=self.population_size, replace=False, p=probas
+ )
+
+ def _gen_short_term_reflection_prompt(
+ self, prompt1: Prompt, prompt2: Prompt
+ ) -> Tuple[str, Prompt, Prompt]:
+ """Generates short-term reflection request into model.
+
+ Args:
+ prompt1 (Prompt): first prompt.
+ prompt2 (Prompt): second prompt.
+
+ Returns:
+ Tuple[str, Prompt, Prompt]:
+ string request, worse prompt, better prompt.
+ """
+ if prompt1.score > prompt2.score:
+ better_prompt, worse_prompt = prompt1, prompt2
+ else:
+ better_prompt, worse_prompt = prompt2, prompt1
+
+ fmt_kwargs = {
+ "PROBLEM_DESCRIPTION": self.problem_description,
+ "WORSE_PROMPT_ROLE": worse_prompt.role,
+ "WORSE_PROMPT_TEXT": worse_prompt.text,
+ "BETTER_PROMPT_ROLE": better_prompt.role,
+ "BETTER_PROMPT_TEXT": better_prompt.text,
+ "WORSE_SCORE": self._format_score(worse_prompt.score),
+ "BETTER_SCORE": self._format_score(better_prompt.score),
+ }
+ if self.evolve_constraints or self._constraints_only:
+ fmt_kwargs["WORSE_PROMPT_CONSTRAINTS"] = worse_prompt.constraints
+ fmt_kwargs["BETTER_PROMPT_CONSTRAINTS"] = better_prompt.constraints
+ if self._role_only or self._constraints_only:
+ fmt_kwargs["FROZEN_PROMPT_TEXT"] = self.initial_prompt
+ fmt_kwargs["FROZEN_PROMPT_ROLE"] = self.initial_role or ""
+ request = self._short_term_template.format(**fmt_kwargs)
+
+ return request, worse_prompt, better_prompt
+
+ def _make_output_path(self, filename: str) -> os.PathLike:
+ """Creates full path for logging based on current iteration.
+
+ Args:
+ filename (str): the file name to save.
+
+ Returns:
+ os.PathLike: final path to save.
+ """
+ return os.path.join(
+ self.output_path, f"Iteration{self.iteration}", f"{filename}.yaml"
+ )
+
+ def _short_term_reflection(
+ self,
+ population: list[Prompt],
+ ) -> Tuple[List[str], List[Prompt], List[Prompt]]:
+ """Short-term reflection before crossovering two individuals.
+
+ Args:
+ population (list[Prompt]): parenting population.
+
+ Returns:
+ Tuple[List[str], List[Prompt], List[Prompt]]:
+ generated short-term hints,
+ worse prompts,
+ better prompts.
+ """
+ requests = []
+ worse_prompts = []
+ better_prompts = []
+ for i in range(0, len(population), 2):
+ parent_1 = population[i]
+ parent_2 = population[i + 1]
+
+ request, worse_p, better_p = self._gen_short_term_reflection_prompt(
+ parent_1, parent_2
+ )
+ requests.append(request)
+ worse_prompts.append(worse_p)
+ better_prompts.append(better_p)
+
+ responses = self._llm_query(requests)
+ responses = [
+ extract_answer(response, self.HINT_TAGS, format_mismatch_label="")
+ for response in responses
+ ]
+ return responses, worse_prompts, better_prompts
+
+ def _crossover(
+ self,
+ short_term_reflection_tuple: Tuple[
+ List[str], List[Prompt], List[Prompt]
+ ],
+ ) -> List[Prompt]:
+ """Provides crossover operation.
+
+ Args:
+ short_term_reflection_tuple
+ (Tuple[List[str], List[Prompt], List[Prompt]]):
+ outputs of short-term reflection.
+
+ Returns:
+ List[Prompt]: new crossed prompts population.
+ """
+ reflection_contents, worse_prompts, better_prompts = (
+ short_term_reflection_tuple
+ )
+ requests = []
+ for reflection, worse_p, better_p in zip(
+ reflection_contents, worse_prompts, better_prompts
+ ):
+ fmt_kwargs = {
+ "PROBLEM_DESCRIPTION": self.problem_description,
+ "WORSE_PROMPT_ROLE": worse_p.role,
+ "WORSE_PROMPT_TEXT": worse_p.text,
+ "BETTER_PROMPT_ROLE": better_p.role,
+ "BETTER_PROMPT_TEXT": better_p.text,
+ "SHORT_TERM_REFLECTION": reflection,
+ "WORSE_SCORE": self._format_score(worse_p.score),
+ "BETTER_SCORE": self._format_score(better_p.score),
+ }
+ if self.evolve_constraints or self._constraints_only:
+ fmt_kwargs["WORSE_PROMPT_CONSTRAINTS"] = worse_p.constraints
+ fmt_kwargs["BETTER_PROMPT_CONSTRAINTS"] = better_p.constraints
+ if self._role_only or self._constraints_only:
+ fmt_kwargs["FROZEN_PROMPT_TEXT"] = self.initial_prompt
+ fmt_kwargs["FROZEN_PROMPT_ROLE"] = self.initial_role or ""
+ request = self._crossover_template.format(**fmt_kwargs)
+ requests.append(request)
+
+ responses = self._llm_query(requests)
+ crossed_population = []
+ for i, response in enumerate(responses):
+ extracted = extract_json(response)
+ if extracted is None:
+ extracted = {}
+
+ if self._role_only:
+ role = extracted.get(
+ "system_behavior", extracted.get("role", "")
+ )
+ text = self.initial_prompt
+ constraints = ""
+ elif self._constraints_only:
+ role = self.initial_role or ""
+ text = self.initial_prompt
+ constraints = extracted.get("output_constraints", "")
+ else:
+ if self.evolve_role:
+ role = extracted.get(
+ "system_behavior", extracted.get("role", "")
+ )
+ else:
+ better_p = better_prompts[i]
+ role = (
+ better_p.role
+ if better_p.role
+ else self.initial_role or ""
+ )
+ text = extracted.get(
+ "task_description",
+ extracted.get(
+ "prompt",
+ extract_answer(
+ response,
+ self.PROMPT_TAGS,
+ format_mismatch_label="",
+ ),
+ ),
+ )
+ constraints = (
+ extracted.get("output_constraints", "")
+ if self.evolve_constraints
+ else ""
+ )
+ crossed_population.append(
+ Prompt(text=text, role=role, constraints=constraints)
+ )
+
+ assert len(crossed_population) == self.population_size
+ return crossed_population
+
+ def _update_elitist(self, population: List[Prompt]) -> None:
+ scores = [prompt.score for prompt in population]
+ best_score, best_sample_idx = max(scores), np.argmax(np.array(scores))
+
+ if (
+ self.best_score_overall is None
+ or best_score >= self.best_score_overall
+ ):
+ self.best_score_overall = best_score
+ self.best_prompt_overall = population[best_sample_idx].text
+ self.best_constraints_overall = population[
+ best_sample_idx
+ ].constraints
+ self.elitist = population[best_sample_idx]
+ logger.info(f"""Iteration {self.iteration}
+ Elitist score: {self.best_score_overall}""")
+ logger.debug(f"Elitist text:\n{self.elitist.text}")
+
+ def _update_iter(self, population: List[Prompt]) -> None:
+ """Updates iteration. Cache current state.
+ Also tracks elitist freeze: if the elitist role has not changed
+ for ELITIST_MAX_FREEZE consecutive epochs, forces the best
+ candidate with a different role to become the new elitist.
+
+ Args:
+ population (List[Prompt]): current population.
+ """
+ logger.info(f"Iteration {self.iteration} finished...")
+ logger.info(f"Best score: {self.best_score_overall}")
+
+ if self.use_enhancements:
+ current_role = self.elitist.role if self.elitist else ""
+ if current_role == self._prev_elitist_role:
+ self._elitist_freeze_count += 1
+ else:
+ self._elitist_freeze_count = 0
+ self._prev_elitist_role = current_role
+
+ if self._elitist_freeze_count >= self.ELITIST_MAX_FREEZE:
+ diverse = [
+ p
+ for p in population
+ if p.role != current_role and p.score is not None
+ ]
+ if diverse:
+ best_diverse = max(diverse, key=lambda p: p.score)
+ logger.debug(
+ f"Elitist frozen {self._elitist_freeze_count} epochs, "
+ f"forcing diverse candidate: '{best_diverse.role[:60]}'"
+ )
+ self.elitist = best_diverse
+ self._prev_elitist_role = best_diverse.role
+ self._elitist_freeze_count = 0
+
+ population = self._reranking(population)
+ self._cache_population(population, self._make_output_path("population"))
+
+ self.iteration += 1
+
+ def _long_term_reflection(self, short_term_reflections: List[str]) -> None:
+ """Long-term reflection before mutation.
+
+ Args:
+ short_term_reflections (List[str]): short-term reflections.
+ """
+ long_term_kwargs = dict(
+ PROBLEM_DESCRIPTION=self.problem_description,
+ PRIOR_LONG_TERM_REFLECTION=self._long_term_reflection_str,
+ NEW_SHORT_TERM_REFLECTIONS="\n".join(short_term_reflections),
+ )
+ if (
+ self._role_only
+ or self._constraints_only
+ or self.text_only
+ or self.evolve_role
+ ):
+ long_term_kwargs["TOP_PROMPTS_HISTORY"] = (
+ self._format_top_prompts_history()
+ )
+ if self._constraints_only:
+ long_term_kwargs["FROZEN_PROMPT_TEXT"] = self.initial_prompt
+ long_term_kwargs["FROZEN_PROMPT_ROLE"] = self.initial_role or ""
+ request = self._long_term_template.format(**long_term_kwargs)
+
+ response = self._llm_query([request])[0]
+
+ self._long_term_reflection_str = extract_answer(
+ response, self.HINT_TAGS, format_mismatch_label=""
+ )
+
+ def _llm_query(self, requests: List[str]) -> List[str]:
+ """Provides api to query requests to the model.
+ Retries up to 3 times with exponential backoff on failure.
+
+ Args:
+ requests (List[str]): string requests.
+
+ Returns:
+ List[str]: model answers.
+ """
+ for attempt in range(3):
+ try:
+ answers = self.model.batch(requests)
+ return [
+ a.content if isinstance(a, AIMessage) else a
+ for a in answers
+ ]
+ except Exception as e:
+ if attempt < 2:
+ logger.warning(
+ f"LLM query failed (attempt {attempt + 1}): {e}. Retrying..."
+ )
+ time.sleep(5 * (attempt + 1))
+ else:
+ raise
+
+ def _mutate(self) -> List[Prompt]:
+ """Elitist-based mutation.
+
+ Returns:
+ List[Prompt]: generated population.
+ """
+ fmt_kwargs = {
+ "PROBLEM_DESCRIPTION": self.problem_description,
+ "LONG_TERM_REFLECTION": self._long_term_reflection_str,
+ "ELITIST_PROMPT_ROLE": self.elitist.role,
+ "ELITIST_PROMPT_TEXT": self.elitist.text,
+ "ELITIST_SCORE": self._format_score(self.elitist.score),
+ }
+ if self.evolve_constraints or self._constraints_only:
+ fmt_kwargs["ELITIST_PROMPT_CONSTRAINTS"] = self.elitist.constraints
+ if (
+ self._role_only
+ or self._constraints_only
+ or self.text_only
+ or self.evolve_role
+ ):
+ fmt_kwargs["BAD_EXAMPLES"] = self._format_bad_examples()
+ if self._role_only or self._constraints_only:
+ fmt_kwargs["FROZEN_PROMPT_TEXT"] = self.initial_prompt
+ fmt_kwargs["FROZEN_PROMPT_ROLE"] = self.initial_role or ""
+ request = self._mutation_template.format(**fmt_kwargs)
+ responses = self._llm_query([request] * self.population_size)
+ mutated_population = []
+ fixed_role = (
+ self.elitist.role if self.elitist and not self.evolve_role else None
+ )
+ if fixed_role is None and not self.evolve_role:
+ fixed_role = self.initial_role or ""
+
+ for response in responses:
+ extracted = extract_json(response)
+ if extracted is None:
+ extracted = {}
+
+ if self._role_only:
+ role = extracted.get(
+ "system_behavior", extracted.get("role", "")
+ )
+ text = self.initial_prompt
+ constraints = ""
+ elif self._constraints_only:
+ role = self.initial_role or ""
+ text = self.initial_prompt
+ constraints = extracted.get("output_constraints", "")
+ else:
+ if self.evolve_role:
+ role = extracted.get(
+ "system_behavior", extracted.get("role", "")
+ )
+ else:
+ role = fixed_role
+ text = extracted.get(
+ "task_description",
+ extracted.get(
+ "prompt",
+ extract_answer(
+ response,
+ self.PROMPT_TAGS,
+ format_mismatch_label="",
+ ),
+ ),
+ )
+ constraints = (
+ extracted.get("output_constraints", "")
+ if self.evolve_constraints
+ else ""
+ )
+ mutated_population.append(
+ Prompt(
+ text=text,
+ role=role,
+ constraints=constraints,
+ origin=PromptOrigin.MUTATED,
+ )
+ )
+ return mutated_population
+
+ def evolution(self, skip_validation: bool = False) -> str:
+ """Provides evolution operation.
+
+ Selection -> Short-term reflection -> Long-term reflection
+ -> Elitist-based mutation -> Survival.
+
+ After all self.num_epochs epochs the best three prompts are selected.
+ They will be evaluated on test split of dataset then.
+ And based on their test scores,
+ the best prompt will be returned.
+
+ Returns:
+ str: best evoluted prompt
+ """
+
+ population = np.array(self._init_pop())
+ self._cache_population(
+ population, self._make_output_path("initial_population.yaml")
+ )
+
+ while self.iteration < self.num_epochs:
+ parent_population = self._selection(population)
+
+ short_term_reflection_tuple = self._short_term_reflection(
+ parent_population
+ )
+ self._cache_data(
+ short_term_reflection_tuple[0],
+ self._make_output_path("short_term_reflections"),
+ )
+
+ crossed_population = self._crossover(short_term_reflection_tuple)
+
+ self._evaluation(crossed_population)
+ self._update_elitist(crossed_population)
+
+ self._long_term_reflection(short_term_reflection_tuple[0])
+ self._cache_data(
+ self._long_term_reflection_str,
+ self._make_output_path("long_term_reflection"),
+ )
+
+ mutated_population = self._mutate()
+ self._evaluation(mutated_population)
+
+ population = np.append(population, np.array(crossed_population))
+ population = np.append(population, np.array(mutated_population))
+ self._update_elitist(population)
+ population = self._survive(population, temperature=1e-1)
+
+ if self.elitist is not None and self.elitist not in population:
+ logger.debug("Elitist should always live")
+ population = np.append(population, np.array([self.elitist]))
+
+ if self.use_enhancements:
+ self._update_hall_of_fame(population)
+ self._cache_data(
+ self._elitist_bad_examples,
+ self._make_output_path("bad_examples"),
+ )
+ self._cache_data(
+ [
+ {
+ "score": self._format_score(p.score),
+ "text": p.text,
+ "role": p.role,
+ "constraints": p.constraints,
+ }
+ for p in self._hall_of_fame[:5]
+ ],
+ self._make_output_path("top_prompts_history"),
+ )
+ self._update_iter(population)
+
+ logger.info(f"BEST TRAIN SCORE: {self.best_score_overall}")
+
+ population = self._reranking(population)
+ final_candidates = list(population[:3])
+ if self.elitist is not None:
+ if not any(
+ c.text == self.elitist.text
+ and c.role == self.elitist.role
+ and c.constraints == self.elitist.constraints
+ for c in final_candidates
+ ):
+ final_candidates.append(self.elitist)
+
+ if self.use_enhancements:
+ seen = {(c.text, c.role, c.constraints) for c in final_candidates}
+ for hof_p in self._hall_of_fame:
+ if (hof_p.text, hof_p.role, hof_p.constraints) not in seen:
+ final_candidates.append(hof_p)
+ seen.add((hof_p.text, hof_p.role, hof_p.constraints))
+ if len(final_candidates) >= 6:
+ break
+
+ if not skip_validation:
+ logger.info(
+ f"Final validation: {len(final_candidates)} candidates "
+ f"({'with HoF' if self.use_enhancements else 'no HoF'})"
+ )
+ final_candidates = np.array(final_candidates)
+ self._evaluation(final_candidates, split="validation")
+ final_candidates = self._reranking(final_candidates)
+ self._cache_population(
+ final_candidates,
+ self._make_output_path("best_prompts_infer.yaml"),
+ )
+ self.elitist = final_candidates[0]
+ self.best_prompt_overall = self.elitist.text
+ self.best_role_overall = self.elitist.role
+ self.best_constraints_overall = self.elitist.constraints
+ self.best_score_overall = self.elitist.score
+ logger.info(f"BEST VALIDATION SCORE: {self.best_score_overall}")
+ logger.debug(f"BEST ROLE:\n{self.best_role_overall}")
+ logger.debug(f"BEST PROMPT:\n{self.best_prompt_overall}")
+ if self.best_constraints_overall:
+ logger.debug(
+ f"BEST CONSTRAINTS:\n{self.best_constraints_overall}"
+ )
+ else:
+ logger.info("Skipping final validation (intermediate phase).")
+ if self.elitist is not None:
+ self.best_prompt_overall = self.elitist.text
+ self.best_role_overall = self.elitist.role
+ self.best_constraints_overall = self.elitist.constraints
+ logger.info(f"BEST TRAIN SCORE (kept): {self.best_score_overall}")
+
+ return self.best_prompt_overall
diff --git a/coolprompt/optimizer/reflective_prompt/coevo_evoluter.py b/coolprompt/optimizer/reflective_prompt/coevo_evoluter.py
new file mode 100644
index 00000000..f836954b
--- /dev/null
+++ b/coolprompt/optimizer/reflective_prompt/coevo_evoluter.py
@@ -0,0 +1,520 @@
+import re
+from typing import Dict, List, Optional, Tuple
+
+from pydantic import (
+ BaseModel,
+ ValidationError,
+ field_validator,
+ model_validator,
+)
+from langchain_core.language_models.base import BaseLanguageModel
+
+from coolprompt.evaluator import Evaluator
+from coolprompt.optimizer.reflective_prompt.coevo_base_evoluter import (
+ ReflectiveEvoluter,
+)
+from coolprompt.optimizer.reflective_prompt.prompt import Prompt, PromptOrigin
+from coolprompt.utils.logging_config import logger
+from coolprompt.utils.parsing import extract_json, extract_answer
+from coolprompt.utils.prompt_templates.reflective_templates_coevo_enhanced import (
+ PARAPHRASING_TEMPLATE_COEVO_ENH,
+ SHORT_TERM_REFLECTION_TEMPLATE_COEVO_ENH,
+ LONG_TERM_REFLECTION_TEMPLATE_COEVO_ENH,
+ CROSSOVER_TEMPLATE_COEVO_ENH,
+ MUTATION_TEMPLATE_COEVO_ENH,
+ PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_ENH,
+)
+from coolprompt.utils.prompt_templates.reflective_templates_coevo_per_field import (
+ PARAPHRASING_TEMPLATE_COEVO_PF,
+ SHORT_TERM_REFLECTION_TEMPLATE_COEVO_PF,
+ LONG_TERM_REFLECTION_TEMPLATE_COEVO_PF,
+ CROSSOVER_TEMPLATE_COEVO_PF,
+ MUTATION_TEMPLATE_COEVO_PF,
+ PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_PF,
+)
+
+
+def _sanitize(value: str) -> str:
+ value = value.strip().strip('"').strip("'").strip()
+ value = re.sub(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f\u2028\u2029]", "", value)
+ return value
+
+
+class _ThreeFieldOutput(BaseModel):
+ task_description: str = ""
+ system_behavior: str = ""
+ output_constraints: str = ""
+
+ @field_validator(
+ "task_description",
+ "system_behavior",
+ "output_constraints",
+ mode="before",
+ )
+ @classmethod
+ def clean_field(cls, v):
+ return _sanitize(str(v)) if v else ""
+
+ @model_validator(mode="after")
+ def check_task_not_empty(self):
+ if not self.task_description:
+ raise ValueError("task_description is empty")
+ return self
+
+
+class CoevoEvoluter(ReflectiveEvoluter):
+ """Evoluter that coevolves all three prompt fields simultaneously.
+
+ Optimizes task_description, system_behavior and output_constraints
+ together in each epoch. Uses Pydantic to validate LLM outputs and
+ runs field ablation at the end to pick the best field combination.
+ """
+
+ def __init__(
+ self,
+ model: BaseLanguageModel,
+ evaluator: Evaluator,
+ train_dataset: List[str],
+ train_targets: List[str],
+ validation_dataset: List[str],
+ validation_targets: List[str],
+ problem_description: str,
+ initial_prompt: Optional[str] = None,
+ initial_role: Optional[str] = None,
+ initial_constraints: Optional[str] = None,
+ population_size: int = 10,
+ num_epochs: int = 10,
+ output_path: str = "./coevo_outputs",
+ use_cache: bool = True,
+ use_enhancements: bool = True,
+ use_bad_examples: Optional[bool] = None,
+ val_evaluator: Optional[Evaluator] = None,
+ ) -> None:
+ super().__init__(
+ model=model,
+ evaluator=evaluator,
+ train_dataset=train_dataset,
+ train_targets=train_targets,
+ validation_dataset=validation_dataset,
+ validation_targets=validation_targets,
+ problem_description=problem_description,
+ initial_prompt=initial_prompt,
+ initial_role=initial_role,
+ initial_constraints=initial_constraints,
+ evolve_role=True,
+ evolve_constraints=True,
+ population_size=population_size,
+ num_epochs=num_epochs,
+ output_path=output_path,
+ use_cache=use_cache,
+ use_enhancements=use_enhancements,
+ use_bad_examples=use_bad_examples,
+ freeze_text=False,
+ text_only=False,
+ val_evaluator=val_evaluator,
+ )
+ self.candidates: List[Dict] = []
+
+ self._paraphrasing_template = PARAPHRASING_TEMPLATE_COEVO_ENH
+ self._crossover_template = CROSSOVER_TEMPLATE_COEVO_ENH
+ self._mutation_template = MUTATION_TEMPLATE_COEVO_ENH
+ self._short_term_template = SHORT_TERM_REFLECTION_TEMPLATE_COEVO_ENH
+ self._long_term_template = LONG_TERM_REFLECTION_TEMPLATE_COEVO_ENH
+ self._initial_prompt_template = PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_ENH
+
+ def _llm_query(self, requests: List[str]) -> List[str]:
+ results = []
+ for req in requests:
+ results.extend(super()._llm_query([_sanitize(req)]))
+ return results
+
+ def _parse_3f_response(
+ self,
+ response: str,
+ fallback_text: str = "",
+ fallback_role: str = "",
+ fallback_constraints: str = "",
+ ) -> Dict[str, str]:
+ raw = extract_json(response) or {}
+
+ try:
+ parsed = _ThreeFieldOutput(
+ task_description=raw.get("task_description") or fallback_text,
+ system_behavior=raw.get("system_behavior") or fallback_role,
+ output_constraints=raw.get("output_constraints")
+ or fallback_constraints,
+ )
+ except (ValidationError, ValueError) as e:
+ logger.warning(
+ f"_parse_3f_response validation failed ({e}), using fallback"
+ )
+ return {
+ "task_description": _sanitize(fallback_text),
+ "system_behavior": _sanitize(fallback_role),
+ "output_constraints": _sanitize(fallback_constraints),
+ }
+ return {
+ "task_description": parsed.task_description,
+ "system_behavior": parsed.system_behavior,
+ "output_constraints": parsed.output_constraints,
+ }
+
+ def _crossover(
+ self,
+ short_term_reflection_tuple: Tuple[
+ List[str], List[Prompt], List[Prompt]
+ ],
+ ) -> List[Prompt]:
+ reflection_contents, worse_prompts, better_prompts = (
+ short_term_reflection_tuple
+ )
+ requests = []
+ for reflection, worse_p, better_p in zip(
+ reflection_contents, worse_prompts, better_prompts
+ ):
+ request = self._crossover_template.format(
+ PROBLEM_DESCRIPTION=self.problem_description,
+ WORSE_PROMPT_TEXT=worse_p.text,
+ WORSE_PROMPT_ROLE=worse_p.role,
+ WORSE_PROMPT_CONSTRAINTS=worse_p.constraints,
+ BETTER_PROMPT_TEXT=better_p.text,
+ BETTER_PROMPT_ROLE=better_p.role,
+ BETTER_PROMPT_CONSTRAINTS=better_p.constraints,
+ SHORT_TERM_REFLECTION=reflection,
+ WORSE_SCORE=self._format_score(worse_p.score),
+ BETTER_SCORE=self._format_score(better_p.score),
+ )
+ requests.append(request)
+
+ responses = self._llm_query(requests)
+ crossed_population = []
+ for i, response in enumerate(responses):
+ fields = self._parse_3f_response(
+ response,
+ fallback_text=better_prompts[i].text,
+ fallback_role=better_prompts[i].role,
+ fallback_constraints=better_prompts[i].constraints,
+ )
+ crossed_population.append(
+ Prompt(
+ text=fields["task_description"],
+ role=fields["system_behavior"],
+ constraints=fields["output_constraints"],
+ origin=PromptOrigin.EVOLUTED,
+ )
+ )
+
+ assert len(crossed_population) == self.population_size
+ return crossed_population
+
+ def _mutate(self) -> List[Prompt]:
+ request = self._mutation_template.format(
+ PROBLEM_DESCRIPTION=self.problem_description,
+ LONG_TERM_REFLECTION=self._long_term_reflection_str,
+ ELITIST_PROMPT_TEXT=self.elitist.text,
+ ELITIST_PROMPT_ROLE=self.elitist.role,
+ ELITIST_PROMPT_CONSTRAINTS=self.elitist.constraints,
+ ELITIST_SCORE=self._format_score(self.elitist.score),
+ BAD_EXAMPLES=self._format_bad_examples(),
+ )
+ responses = self._llm_query([request] * self.population_size)
+ mutated_population = []
+ for response in responses:
+ fields = self._parse_3f_response(
+ response,
+ fallback_text=self.elitist.text,
+ fallback_role=self.elitist.role,
+ fallback_constraints=self.elitist.constraints,
+ )
+ mutated_population.append(
+ Prompt(
+ text=fields["task_description"],
+ role=fields["system_behavior"],
+ constraints=fields["output_constraints"],
+ origin=PromptOrigin.MUTATED,
+ )
+ )
+ return mutated_population
+
+ def _eval_val(self, prompt: str, role: str, constraints: str) -> float:
+ result = self.val_evaluator.evaluate(
+ prompt=prompt,
+ dataset=self.validation_dataset,
+ targets=self.validation_targets,
+ system_role=role or None,
+ constraints=constraints or None,
+ )
+ assert isinstance(result, float)
+ return result
+
+ def _field_ablation(
+ self,
+ best_text: str,
+ best_role: str,
+ best_constraints: str,
+ score_text_role_constraints: Optional[float] = None,
+ ) -> Tuple[List[Dict], Dict]:
+ logger.info(
+ "[Field Ablation] Evaluating field combinations on validation set..."
+ )
+ score_a = self._eval_val(best_text, "", "")
+ logger.info(f" text_only: {score_a:.4f}")
+ score_b = self._eval_val(best_text, best_role, "")
+ logger.info(f" text_role: {score_b:.4f}")
+
+ candidates = [
+ {
+ "combo": "text_only",
+ "prompt": best_text,
+ "role": "",
+ "constraints": "",
+ "val_score": score_a,
+ },
+ {
+ "combo": "text_role",
+ "prompt": best_text,
+ "role": best_role,
+ "constraints": "",
+ "val_score": score_b,
+ },
+ ]
+
+ if best_constraints:
+ if score_text_role_constraints is None:
+ score_text_role_constraints = self._eval_val(
+ best_text, best_role, best_constraints
+ )
+ logger.info(
+ f" text_role_constraints: {score_text_role_constraints:.4f}"
+ )
+ candidates.append(
+ {
+ "combo": "text_role_constraints",
+ "prompt": best_text,
+ "role": best_role,
+ "constraints": best_constraints,
+ "val_score": score_text_role_constraints,
+ }
+ )
+
+ _combo_order = {
+ "text_only": 0,
+ "text_role": 1,
+ "text_role_constraints": 2,
+ }
+ best_c = max(
+ candidates, key=lambda c: (c["val_score"], _combo_order[c["combo"]])
+ )
+ logger.info(
+ f"Best combo: {best_c['combo']} (val={best_c['val_score']:.4f})"
+ )
+ return candidates, best_c
+
+ def evolution(self) -> Optional[str]:
+ super().evolution()
+
+ if self.best_prompt_overall:
+ candidates, best_c = self._field_ablation(
+ best_text=self.best_prompt_overall,
+ best_role=self.best_role_overall or "",
+ best_constraints=self.best_constraints_overall or "",
+ score_text_role_constraints=self.best_score_overall,
+ )
+ self.candidates = candidates
+ self.best_prompt_overall = best_c["prompt"]
+ self.best_role_overall = best_c["role"]
+ self.best_constraints_overall = best_c["constraints"]
+ self.best_score_overall = best_c["val_score"]
+ logger.info(
+ f"Field ablation done. Best combo: {best_c['combo']} "
+ f"(val={best_c['val_score']:.4f})"
+ )
+
+ return self.best_prompt_overall
+
+
+class PerFieldCoevoEvoluter(CoevoEvoluter):
+ """CoevoEvoluter variant that uses per-field reflection hints.
+
+ Each reflection step produces three separate hints β one each for
+ task_description, system_behavior, and output_constraints β instead of a
+ single combined hint. Crossover and mutation templates consume these
+ field-specific hints directly.
+
+ All other enhancements (HoF, role length penalty, similarity penalty,
+ field ablation) are inherited from CoevoEvoluter unchanged.
+ """
+
+ HINT_TASK_TAGS = ("", "")
+ HINT_ROLE_TAGS = ("", "")
+ HINT_CONSTRAINTS_TAGS = ("", "")
+
+ _FALLBACK_HINT = "(no hint)"
+
+ def __init__(self, *args, **kwargs):
+ super().__init__(*args, **kwargs)
+ self._paraphrasing_template = PARAPHRASING_TEMPLATE_COEVO_PF
+ self._crossover_template = CROSSOVER_TEMPLATE_COEVO_PF
+ self._mutation_template = MUTATION_TEMPLATE_COEVO_PF
+ self._short_term_template = SHORT_TERM_REFLECTION_TEMPLATE_COEVO_PF
+ self._long_term_template = LONG_TERM_REFLECTION_TEMPLATE_COEVO_PF
+ self._initial_prompt_template = PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_PF
+
+ self._per_field_short_hints: List[Dict[str, str]] = []
+ self._per_field_long_hints: Dict[str, str] = {
+ "task": self._FALLBACK_HINT,
+ "role": self._FALLBACK_HINT,
+ "constraints": self._FALLBACK_HINT,
+ }
+
+ def _parse_per_field_hints(self, response: str) -> Dict[str, str]:
+ task = extract_answer(
+ response, self.HINT_TASK_TAGS, format_mismatch_label=""
+ ).strip()
+ role = extract_answer(
+ response, self.HINT_ROLE_TAGS, format_mismatch_label=""
+ ).strip()
+ constraints = extract_answer(
+ response, self.HINT_CONSTRAINTS_TAGS, format_mismatch_label=""
+ ).strip()
+ return {
+ "task": task or self._FALLBACK_HINT,
+ "role": role or self._FALLBACK_HINT,
+ "constraints": constraints or self._FALLBACK_HINT,
+ }
+
+ def _short_term_reflection(self, population):
+ requests = []
+ worse_prompts = []
+ better_prompts = []
+ for i in range(0, len(population), 2):
+ parent_1 = population[i]
+ parent_2 = population[i + 1]
+ request, worse_p, better_p = self._gen_short_term_reflection_prompt(
+ parent_1, parent_2
+ )
+ requests.append(request)
+ worse_prompts.append(worse_p)
+ better_prompts.append(better_p)
+
+ responses = self._llm_query(requests)
+
+ self._per_field_short_hints = []
+ combined_strings = []
+ for response in responses:
+ hints = self._parse_per_field_hints(response)
+ self._per_field_short_hints.append(hints)
+ combined_strings.append(
+ f"task_description: {hints['task']}\n"
+ f"system_behavior: {hints['role']}\n"
+ f"output_constraints: {hints['constraints']}"
+ )
+
+ return combined_strings, worse_prompts, better_prompts
+
+ def _long_term_reflection(self, short_term_reflections: List[str]) -> None:
+ request = self._long_term_template.format(
+ PROBLEM_DESCRIPTION=self.problem_description,
+ TOP_PROMPTS_HISTORY=self._format_top_prompts_history(),
+ PRIOR_TASK_HINT=self._per_field_long_hints["task"],
+ PRIOR_ROLE_HINT=self._per_field_long_hints["role"],
+ PRIOR_CONSTRAINTS_HINT=self._per_field_long_hints["constraints"],
+ NEW_SHORT_TERM_REFLECTIONS="\n---\n".join(short_term_reflections),
+ )
+ response = self._llm_query([request])[0]
+ hints = self._parse_per_field_hints(response)
+ self._per_field_long_hints = hints
+ self._long_term_reflection_str = (
+ f"task_description: {hints['task']}\n"
+ f"system_behavior: {hints['role']}\n"
+ f"output_constraints: {hints['constraints']}"
+ )
+
+ def _crossover(
+ self,
+ short_term_reflection_tuple: Tuple[
+ List[str], List[Prompt], List[Prompt]
+ ],
+ ) -> List[Prompt]:
+ _, worse_prompts, better_prompts = short_term_reflection_tuple
+ requests = []
+ for i, (worse_p, better_p) in enumerate(
+ zip(worse_prompts, better_prompts)
+ ):
+ hints = (
+ self._per_field_short_hints[i]
+ if i < len(self._per_field_short_hints)
+ else {
+ "task": self._FALLBACK_HINT,
+ "role": self._FALLBACK_HINT,
+ "constraints": self._FALLBACK_HINT,
+ }
+ )
+ request = self._crossover_template.format(
+ PROBLEM_DESCRIPTION=self.problem_description,
+ WORSE_PROMPT_TEXT=worse_p.text,
+ WORSE_PROMPT_ROLE=worse_p.role,
+ WORSE_PROMPT_CONSTRAINTS=worse_p.constraints,
+ BETTER_PROMPT_TEXT=better_p.text,
+ BETTER_PROMPT_ROLE=better_p.role,
+ BETTER_PROMPT_CONSTRAINTS=better_p.constraints,
+ TASK_HINT=hints["task"],
+ ROLE_HINT=hints["role"],
+ CONSTRAINTS_HINT=hints["constraints"],
+ WORSE_SCORE=self._format_score(worse_p.score),
+ BETTER_SCORE=self._format_score(better_p.score),
+ )
+ requests.append(request)
+
+ responses = self._llm_query(requests)
+ crossed_population = []
+ for i, response in enumerate(responses):
+ fields = self._parse_3f_response(
+ response,
+ fallback_text=better_prompts[i].text,
+ fallback_role=better_prompts[i].role,
+ fallback_constraints=better_prompts[i].constraints,
+ )
+ crossed_population.append(
+ Prompt(
+ text=fields["task_description"],
+ role=fields["system_behavior"],
+ constraints=fields["output_constraints"],
+ origin=PromptOrigin.EVOLUTED,
+ )
+ )
+
+ assert len(crossed_population) == self.population_size
+ return crossed_population
+
+ def _mutate(self) -> List[Prompt]:
+ hints = self._per_field_long_hints
+ request = self._mutation_template.format(
+ PROBLEM_DESCRIPTION=self.problem_description,
+ TASK_HINT=hints["task"],
+ ROLE_HINT=hints["role"],
+ CONSTRAINTS_HINT=hints["constraints"],
+ ELITIST_PROMPT_TEXT=self.elitist.text,
+ ELITIST_PROMPT_ROLE=self.elitist.role,
+ ELITIST_PROMPT_CONSTRAINTS=self.elitist.constraints,
+ ELITIST_SCORE=self._format_score(self.elitist.score),
+ BAD_EXAMPLES=self._format_bad_examples(),
+ )
+ responses = self._llm_query([request] * self.population_size)
+ mutated_population = []
+ for response in responses:
+ fields = self._parse_3f_response(
+ response,
+ fallback_text=self.elitist.text,
+ fallback_role=self.elitist.role,
+ fallback_constraints=self.elitist.constraints,
+ )
+ mutated_population.append(
+ Prompt(
+ text=fields["task_description"],
+ role=fields["system_behavior"],
+ constraints=fields["output_constraints"],
+ origin=PromptOrigin.MUTATED,
+ )
+ )
+ return mutated_population
diff --git a/coolprompt/optimizer/reflective_prompt/factorized_evoluter.py b/coolprompt/optimizer/reflective_prompt/factorized_evoluter.py
new file mode 100644
index 00000000..cc380f8d
--- /dev/null
+++ b/coolprompt/optimizer/reflective_prompt/factorized_evoluter.py
@@ -0,0 +1,349 @@
+import os
+from typing import List, Optional, Tuple
+
+from langchain_core.language_models.base import BaseLanguageModel
+from langchain_core.messages.ai import AIMessage
+
+from coolprompt.evaluator import Evaluator
+from coolprompt.optimizer.reflective_prompt.coevo_base_evoluter import (
+ ReflectiveEvoluter,
+)
+from coolprompt.utils.logging_config import logger
+from coolprompt.utils.parsing import extract_json
+from coolprompt.utils.prompt_templates.reflective_templates_factorized import (
+ DEDUP_ROLE_TEMPLATE,
+ DEDUP_CONSTRAINTS_TEMPLATE,
+)
+
+
+class FactorizedEvoluter:
+
+ def __init__(
+ self,
+ model: BaseLanguageModel,
+ evaluator: Evaluator,
+ train_dataset: List[str],
+ train_targets: List[str],
+ validation_dataset: List[str],
+ validation_targets: List[str],
+ problem_description: str,
+ initial_prompt: Optional[str] = None,
+ initial_role: Optional[str] = None,
+ initial_constraints: Optional[str] = None,
+ population_size: int = 5,
+ phase_epochs: Tuple[int, int, int] = (4, 3, 3),
+ run_constraints_phase: bool = True,
+ output_path: str = "./factorized_outputs",
+ use_cache: bool = True,
+ use_enhancements: bool = True,
+ use_dedup: bool = True,
+ val_evaluator: Optional[Evaluator] = None,
+ ) -> None:
+ self.model = model
+ self.evaluator = evaluator
+ self.val_evaluator = val_evaluator or evaluator
+ self.train_dataset = train_dataset
+ self.train_targets = train_targets
+ self.validation_dataset = validation_dataset
+ self.validation_targets = validation_targets
+ self.problem_description = problem_description
+ self.initial_prompt = initial_prompt
+ self.initial_role = initial_role or ""
+ self.initial_constraints = initial_constraints or ""
+ self.population_size = population_size
+ self.phase_epochs = phase_epochs
+ self.run_constraints_phase = run_constraints_phase and bool(
+ initial_constraints
+ )
+ self.output_path = output_path
+ self.use_cache = use_cache
+ self.use_enhancements = use_enhancements
+ self.use_dedup = use_dedup
+
+ self.best_prompt_overall = None
+ self.best_role_overall = None
+ self.best_constraints_overall = None
+ self.best_score_overall = None
+ self.candidates: List[dict] = []
+
+ def _make_phase_evoluter(
+ self,
+ phase_name: str,
+ initial_prompt: Optional[str],
+ initial_role: Optional[str],
+ initial_constraints: Optional[str],
+ num_epochs: int,
+ evolve_role: bool,
+ evolve_constraints: bool,
+ freeze_text: bool,
+ ) -> ReflectiveEvoluter:
+ text_only = (
+ not evolve_role and not freeze_text and not evolve_constraints
+ )
+ return ReflectiveEvoluter(
+ model=self.model,
+ evaluator=self.evaluator,
+ train_dataset=self.train_dataset,
+ train_targets=self.train_targets,
+ validation_dataset=self.validation_dataset,
+ validation_targets=self.validation_targets,
+ problem_description=self.problem_description,
+ initial_prompt=initial_prompt,
+ initial_role=initial_role,
+ initial_constraints=initial_constraints,
+ evolve_role=evolve_role,
+ evolve_constraints=evolve_constraints,
+ freeze_text=freeze_text,
+ text_only=text_only,
+ population_size=self.population_size,
+ num_epochs=num_epochs,
+ use_cache=self.use_cache,
+ output_path=os.path.join(self.output_path, phase_name),
+ use_enhancements=self.use_enhancements,
+ )
+
+ def _eval_val(self, prompt: str, role: str, constraints: str) -> float:
+ result = self.val_evaluator.evaluate(
+ prompt=prompt,
+ dataset=self.validation_dataset,
+ targets=self.validation_targets,
+ system_role=role or None,
+ constraints=constraints or None,
+ )
+ assert isinstance(result, float)
+ return result
+
+ def _field_ablation(
+ self,
+ best_text: Optional[str],
+ best_role: Optional[str],
+ best_constraints: str,
+ score_text_role_constraints: Optional[float] = None,
+ ) -> Tuple[List[dict], dict]:
+ logger.info(
+ "[Field Ablation] Evaluating all field combinations on validation set..."
+ )
+ assert best_text is not None and best_role is not None
+ score_a = self._eval_val(best_text, "", "")
+ logger.info(f" text_only: {score_a:.4f}")
+ score_b = self._eval_val(best_text, best_role, "")
+ logger.info(f" text_role: {score_b:.4f}")
+ candidates = [
+ {
+ "combo": "text_only",
+ "prompt": best_text,
+ "role": "",
+ "constraints": "",
+ "val_score": score_a,
+ },
+ {
+ "combo": "text_role",
+ "prompt": best_text,
+ "role": best_role,
+ "constraints": "",
+ "val_score": score_b,
+ },
+ ]
+ if best_constraints:
+ if score_text_role_constraints is None:
+ score_text_role_constraints = self._eval_val(
+ best_text, best_role, best_constraints
+ )
+ logger.info(
+ f" text_role_constraints: {score_text_role_constraints:.4f}"
+ )
+ candidates.append(
+ {
+ "combo": "text_role_constraints",
+ "prompt": best_text,
+ "role": best_role,
+ "constraints": best_constraints,
+ "val_score": score_text_role_constraints,
+ }
+ )
+ best_c = max(candidates, key=lambda c: c["val_score"])
+ logger.info(f"Best combo: {best_c['combo']} (val={best_c['val_score']:.4f})")
+ return candidates, best_c
+
+ def _llm_call(self, request: str) -> str:
+ responses = self.model.batch([request])
+ r = responses[0]
+ return r.content if isinstance(r, AIMessage) else r
+
+ def _dedup_role(self, task_text: str, role: str) -> str:
+ if not role or not self.use_dedup:
+ return role
+ try:
+ parsed = extract_json(
+ self._llm_call(
+ DEDUP_ROLE_TEMPLATE.format(TASK=task_text, ROLE=role)
+ )
+ )
+ if parsed and "system_behavior" in parsed:
+ cleaned = str(parsed["system_behavior"]).strip()
+ if cleaned != role:
+ logger.info(
+ f"[Dedup role] seed cleaned: '{role[:80]}' '{cleaned[:80]}'"
+ )
+ return cleaned
+ except Exception as e:
+ logger.warning(f"Dedup role failed: {e}. Using original.")
+ return role
+
+ def _dedup_constraints(
+ self, task_text: str, role: str, constraints: str
+ ) -> str:
+ if not constraints or not self.use_dedup:
+ return constraints
+ try:
+ parsed = extract_json(
+ self._llm_call(
+ DEDUP_CONSTRAINTS_TEMPLATE.format(
+ TASK=task_text,
+ ROLE=role or "(none)",
+ CONSTRAINTS=constraints,
+ )
+ )
+ )
+ if parsed and "output_constraints" in parsed:
+ cleaned = str(parsed["output_constraints"]).strip()
+ if cleaned != constraints:
+ logger.info(
+ f"[Dedup constraints] seed cleaned: '{constraints[:80]}' '{cleaned[:80]}'"
+ )
+ return cleaned
+ except Exception as e:
+ logger.warning(f"Dedup constraints failed: {e}. Using original.")
+ return constraints
+
+ def evolution(self) -> str:
+ last_phase = 3 if self.run_constraints_phase else 2
+
+ logger.info(
+ f"Factorized evolution: {self.phase_epochs[0]} + {self.phase_epochs[1]}"
+ + (
+ f" + {self.phase_epochs[2]} epochs"
+ if self.run_constraints_phase
+ else " epochs"
+ )
+ + f" | phases: text -> role"
+ + (" -> constraints" if self.run_constraints_phase else "")
+ )
+
+ logger.info(
+ f"[Phase 1/{last_phase}] Optimizing task_description ({self.phase_epochs[0]} epochs)"
+ )
+ p1 = self._make_phase_evoluter(
+ phase_name="phase1_text",
+ initial_prompt=self.initial_prompt,
+ initial_role=None,
+ initial_constraints=None,
+ num_epochs=self.phase_epochs[0],
+ evolve_role=False,
+ evolve_constraints=False,
+ freeze_text=False,
+ )
+ p1.evolution(skip_validation=True)
+ best_text = p1.best_prompt_overall
+ assert best_text is not None
+ logger.info(f"Phase 1 best text score (train): {p1.best_score_overall:.4f}")
+ logger.info(f"Phase 1 best text: {best_text[:120]}")
+
+ logger.info(
+ f"[Phase 2/{last_phase}] Optimizing system_behavior ({self.phase_epochs[1]} epochs)"
+ )
+ initial_role_for_p2 = self._dedup_role(
+ best_text, self.initial_role or ""
+ )
+ skip_p2_val = self.run_constraints_phase
+ p2 = self._make_phase_evoluter(
+ phase_name="phase2_role",
+ initial_prompt=best_text,
+ initial_role=initial_role_for_p2,
+ initial_constraints=None,
+ num_epochs=self.phase_epochs[1],
+ evolve_role=True,
+ evolve_constraints=False,
+ freeze_text=True,
+ )
+ p2.evolution(skip_validation=skip_p2_val)
+ best_role = p2.best_role_overall
+ logger.info(f"Phase 2 best role score (train): {p2.best_score_overall:.4f}")
+ logger.info(f"Phase 2 best role: {(best_role or '')[:120]}")
+
+ if not self.run_constraints_phase:
+ self.initial_prompt = p1.initial_prompt
+ self.initial_role = p2.initial_role or ""
+ self.initial_constraints = ""
+ candidates, best_c = self._field_ablation(
+ best_text=best_text,
+ best_role=p2.best_role_overall,
+ best_constraints="",
+ )
+ self.candidates = candidates
+ self.best_prompt_overall = best_c["prompt"]
+ self.best_role_overall = best_c["role"]
+ self.best_constraints_overall = best_c["constraints"]
+ self.best_score_overall = best_c["val_score"]
+ return self.best_prompt_overall
+
+ val_text_only = self._eval_val(best_text, "", "")
+ val_text_role = self._eval_val(best_text, best_role or "", "")
+ logger.info(
+ f"[Pre-Phase 3 check] text_only val: {val_text_only:.4f}, text_role val: {val_text_role:.4f}"
+ )
+ if val_text_only >= val_text_role:
+ logger.info("Role does not improve on validation. Skipping Phase 3.")
+ self.initial_prompt = p1.initial_prompt
+ self.initial_role = p2.initial_role or ""
+ self.initial_constraints = ""
+ candidates, best_c = self._field_ablation(
+ best_text=best_text,
+ best_role=best_role or "",
+ best_constraints="",
+ )
+ self.candidates = candidates
+ self.best_prompt_overall = best_c["prompt"]
+ self.best_role_overall = best_c["role"]
+ self.best_constraints_overall = best_c["constraints"]
+ self.best_score_overall = best_c["val_score"]
+ return self.best_prompt_overall
+
+ logger.info(
+ f"[Phase 3/{last_phase}] Optimizing output_constraints ({self.phase_epochs[2]} epochs)"
+ )
+ initial_constraints_for_p3 = self._dedup_constraints(
+ best_text, best_role or "", self.initial_constraints
+ )
+ if not initial_constraints_for_p3 and self.initial_constraints:
+ logger.info("[Dedup] constraints redundant with task/role, using fallback seed")
+ initial_constraints_for_p3 = "Return only the final answer."
+ p3 = self._make_phase_evoluter(
+ phase_name="phase3_constraints",
+ initial_prompt=best_text,
+ initial_role=best_role or "",
+ initial_constraints=initial_constraints_for_p3,
+ num_epochs=self.phase_epochs[2],
+ evolve_role=False,
+ evolve_constraints=True,
+ freeze_text=False,
+ )
+ p3.evolution(skip_validation=False)
+ logger.info(f"Phase 3 best constraints score: {p3.best_score_overall:.4f}")
+
+ self.initial_prompt = p1.initial_prompt
+ self.initial_role = p2.initial_role or ""
+ self.initial_constraints = p3.initial_constraints or ""
+
+ candidates, best_c = self._field_ablation(
+ best_text=best_text,
+ best_role=p2.best_role_overall,
+ best_constraints=p3.best_constraints_overall or "",
+ score_text_role_constraints=p3.best_score_overall,
+ )
+ self.candidates = candidates
+ self.best_prompt_overall = best_c["prompt"]
+ self.best_role_overall = best_c["role"]
+ self.best_constraints_overall = best_c["constraints"]
+ self.best_score_overall = best_c["val_score"]
+ return self.best_prompt_overall
diff --git a/coolprompt/optimizer/reflective_prompt/prompt.py b/coolprompt/optimizer/reflective_prompt/prompt.py
index d2c231a3..63183c99 100644
--- a/coolprompt/optimizer/reflective_prompt/prompt.py
+++ b/coolprompt/optimizer/reflective_prompt/prompt.py
@@ -33,10 +33,10 @@ class BadExample:
input (str): input of the example.
output (str): model output for the example.
correct (str): correct output of the example.
- """
-
- def __init__(self, input: str, output: str, correct: str):
- self.input = input
+ """
+
+ def __init__(self, input: str, output: str, correct: str):
+ self.input = input
self.output = output
self.correct = correct
@@ -69,15 +69,17 @@ def from_dict(cls: Type["BadExample"], data: dict) -> "BadExample":
)
-class Prompt:
- """Prompt candidate with origin, score, and optional failed examples."""
-
- def __init__(
+class Prompt:
+ """Prompt candidate with origin, score, and optional failed examples."""
+
+ def __init__(
self,
text: str,
origin: PromptOrigin = PromptOrigin.EVOLUTED,
score: float = None,
bad_examples: List[BadExample] = [],
+ role: str = "",
+ constraints: str = "",
) -> None:
"""Prompt class.
@@ -88,12 +90,16 @@ def __init__(
score (float, optional): prompt evaluation score. Defaults to None.
bad_examples (List[BadExample]): a list of
bad examples for the prompt.
+ role (str): system behavior / role for the model (CoEvo). Defaults to "".
+ constraints (str): output format constraints (CoEvo). Defaults to "".
"""
self.text = text
self.origin = origin
self.score = score
self.bad_examples = bad_examples
+ self.role = role
+ self.constraints = constraints
def set_score(self, new_score: float) -> None:
"""Records new prompt evaluation score.
@@ -127,6 +133,10 @@ def to_dict(self) -> dict:
"text": self.text,
"origin": self.origin.name,
}
+ if self.role:
+ result["role"] = self.role
+ if self.constraints:
+ result["constraints"] = self.constraints
if self.score is not None:
result["score"] = self.score
if len(self.bad_examples) > 0:
@@ -159,6 +169,8 @@ def from_dict(
BadExample.from_dict(bad_example_data)
for bad_example_data in data.get("bad_examples", [])
],
+ role=data.get("role", ""),
+ constraints=data.get("constraints", ""),
)
def __str__(self) -> str:
diff --git a/coolprompt/optimizer/reflective_prompt/run.py b/coolprompt/optimizer/reflective_prompt/run.py
index 33adac8b..f85aad30 100644
--- a/coolprompt/optimizer/reflective_prompt/run.py
+++ b/coolprompt/optimizer/reflective_prompt/run.py
@@ -1,4 +1,4 @@
-from typing import List, Tuple, override
+from typing import List, Optional, Tuple, override
from langchain_core.language_models import BaseLanguageModel
@@ -9,6 +9,7 @@
BenchmarkContext,
)
from coolprompt.optimizer.reflective_prompt.evoluter import ReflectiveEvoluter
+from coolprompt.optimizer.reflective_prompt.coevo_evoluter import CoevoEvoluter
from coolprompt.utils.deprecation import warn_deprecated
from coolprompt.utils.logging_config import logger
@@ -75,7 +76,7 @@ def reflectiveprompt(
class ReflectiveMethod(AutoPromptingMethod):
- """Reflective prompting method for autoβprompting."""
+ """Reflective prompting method for auto-prompting."""
def optimize(
self,
@@ -129,3 +130,146 @@ def is_data_driven(self) -> bool:
@override
def name(self) -> str:
return "reflective"
+
+
+def coevo(
+ model: BaseLanguageModel,
+ dataset_split: Tuple[List[str], List[str], List[str], List[str]],
+ evaluator: Evaluator,
+ problem_description: str,
+ initial_prompt: Optional[str] = None,
+ initial_role: Optional[str] = None,
+ initial_constraints: Optional[str] = None,
+ use_enhancements: bool = True,
+ use_bad_examples: Optional[bool] = None,
+ **kwargs,
+) -> dict:
+ """Runs CoevoEvoluter optimization β co-evolves task description, system behavior and output constraints.
+
+ Args:
+ model (BaseLanguageModel): a LLM to use.
+ dataset_split (Tuple[List[str], List[str], List[str], List[str]]):
+ train/valid split of dataset and corresponding targets.
+ evaluator (Evaluator): evaluator to compute metrics.
+ problem_description (str): short description of the task to optimize.
+ initial_prompt (str, optional): initial task description. Defaults to None.
+ initial_role (str, optional): initial system behavior. Defaults to None.
+ initial_constraints (str, optional): initial output constraints. Defaults to None.
+ use_enhancements (bool): whether to use enhanced co-evolution templates. Defaults to True.
+ use_bad_examples (bool, optional): whether to feed systematic error examples into mutation. If None, follows use_enhancements.
+ **kwargs: additional parameters (population_size, num_epochs, output_path, use_cache).
+
+ Returns:
+ dict: best evolved prompt with keys:
+ - task_description (str): goes into the human message.
+ - system_behavior (str): goes into the system message.
+ - output_constraints (str): appended to the human message.
+ """
+ train_dataset, validation_dataset, train_targets, validation_targets = (
+ dataset_split
+ )
+ args = {
+ "population_size": 10,
+ "num_epochs": 5,
+ "output_path": "./coevo_outputs",
+ "use_cache": True,
+ }
+ args.update(kwargs)
+ evoluter = CoevoEvoluter(
+ model=model,
+ evaluator=evaluator,
+ train_dataset=train_dataset,
+ train_targets=train_targets,
+ validation_dataset=validation_dataset,
+ validation_targets=validation_targets,
+ problem_description=problem_description,
+ initial_prompt=initial_prompt,
+ initial_role=initial_role,
+ initial_constraints=initial_constraints,
+ use_enhancements=use_enhancements,
+ use_bad_examples=use_bad_examples,
+ population_size=args["population_size"],
+ num_epochs=args["num_epochs"],
+ output_path=args["output_path"],
+ use_cache=args["use_cache"],
+ )
+ logger.info("Starting CoEvo optimization...")
+ logger.debug(f"Start prompt:\n{initial_prompt}")
+ logger.debug(f"Problem description:\n{problem_description}")
+ evoluter.evolution()
+ logger.info("CoEvo optimization completed")
+ return {
+ "task_description": evoluter.best_prompt_overall or "",
+ "system_behavior": evoluter.best_role_overall or "",
+ "output_constraints": evoluter.best_constraints_overall or "",
+ }
+
+
+class CoevoMethod(AutoPromptingMethod):
+ """Co-evolution method: structured prompt of three fields
+ (task_description, system_behavior, output_constraints).
+
+ ``optimize`` returns the task_description as the main prompt and exposes
+ the evolved role and constraints via ``last_role`` / ``last_constraints``,
+ which PromptTuner surfaces as final_role / final_constraints.
+ """
+
+ last_role: str = ""
+ last_constraints: str = ""
+
+ def optimize(
+ self,
+ model,
+ initial_prompt,
+ dataset_split,
+ evaluator,
+ problem_description,
+ **kwargs,
+ ):
+ """Run CoEvo through the shared method interface."""
+ result = coevo(
+ model=model,
+ dataset_split=dataset_split,
+ evaluator=evaluator,
+ problem_description=problem_description,
+ initial_prompt=initial_prompt,
+ **kwargs,
+ )
+ self.last_role = result["system_behavior"]
+ self.last_constraints = result["output_constraints"]
+ return result["task_description"]
+
+ def run_configured_benchmark(
+ self,
+ ctx: BenchmarkContext,
+ start_prompt: str,
+ ) -> str:
+ """Run CoEvo from a benchmark context."""
+ problem_description = ctx.config.get("problem_description")
+ if problem_description is None:
+ generator = SyntheticDataGenerator(ctx._system_model)
+ problem_description = generator._generate_problem_description(
+ prompt=start_prompt
+ )
+ mc = ctx.config["method"]
+ return self.optimize(
+ ctx.model,
+ start_prompt,
+ dataset_split=ctx.dataset_split,
+ evaluator=ctx.evaluator,
+ problem_description=problem_description,
+ population_size=mc.get("population_size", 10),
+ num_epochs=mc.get("num_epochs", 5),
+ output_path=mc.get("output_path", "./coevo_outputs"),
+ use_cache=mc.get("use_cache", True),
+ use_enhancements=mc.get("use_enhancements", True),
+ use_bad_examples=mc.get("use_bad_examples", None),
+ )
+
+ def is_data_driven(self) -> bool:
+ return True
+
+ @property
+ @override
+ def name(self) -> str:
+ return "coevo"
diff --git a/coolprompt/utils/prompt_templates/reflective_templates_coevo_enhanced.py b/coolprompt/utils/prompt_templates/reflective_templates_coevo_enhanced.py
new file mode 100644
index 00000000..9d7a3645
--- /dev/null
+++ b/coolprompt/utils/prompt_templates/reflective_templates_coevo_enhanced.py
@@ -0,0 +1,149 @@
+PARAPHRASING_TEMPLATE_COEVO_ENH = """Create {NUM_PROMPTS} diverse initial variants of the following three-field configuration.
+
+Task: {PROBLEM_DESCRIPTION}
+
+Seed configuration:
+task_description: {PROMPT}
+system_behavior: {ROLE}
+output_constraints: {CONSTRAINTS}
+
+Rules for each variant:
+- Vary at least two fields meaningfully from the seed.
+- "task_description": change wording, directness, or how the output format is stated β preserve the task intent.
+- "system_behavior": vary the reasoning strategy, cognitive angle, or focus area. Can start "You are [role]" only if immediately followed by a concrete behavioral instruction. 8β25 words. Must NOT restate task content.
+- "output_constraints": vary format rules β length limits, structure, what to include or exclude. Must NOT include reasoning instructions or decision strategies (those belong in system_behavior).
+- Each variant must differ meaningfully from the others.
+
+Output JSON only:
+{{
+ "prompts": [
+ {{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}},
+ {{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}},
+ ...
+ ]
+}}
+Output JSON data only.
+"""
+
+SHORT_TERM_REFLECTION_TEMPLATE_COEVO_ENH = """You are an expert in prompt optimization. Compare two three-field configurations and identify what makes the better one score higher.
+
+Task: {PROBLEM_DESCRIPTION}
+
+[Worse configuration] (score: {WORSE_SCORE})
+task_description: {WORSE_PROMPT_TEXT}
+system_behavior: {WORSE_PROMPT_ROLE}
+output_constraints: {WORSE_PROMPT_CONSTRAINTS}
+
+[Better configuration] (score: {BETTER_SCORE})
+task_description: {BETTER_PROMPT_TEXT}
+system_behavior: {BETTER_PROMPT_ROLE}
+output_constraints: {BETTER_PROMPT_CONSTRAINTS}
+
+Analyze each field separately:
+- task_description: what difference in wording, directness, or format specification matters?
+- system_behavior: what difference in reasoning strategy, focus area, or decision rule matters?
+- output_constraints: what difference in format rule, length limit, or exclusion matters?
+
+Then write ONE combined actionable hint (under 30 words) identifying the most impactful change.
+Wrap the hint with .
+"""
+
+LONG_TERM_REFLECTION_TEMPLATE_COEVO_ENH = """You are an expert in prompt optimization. Synthesize patterns from the best-performing configurations found so far.
+
+Task: {PROBLEM_DESCRIPTION}
+
+Best configurations found so far (ranked by score, best first):
+{TOP_PROMPTS_HISTORY}
+
+Prior accumulated insight:
+{PRIOR_LONG_TERM_REFLECTION}
+
+New per-field observations from recent comparisons:
+{NEW_SHORT_TERM_REFLECTIONS}
+
+Study the top configurations above. Identify what distinguishes the highest-scoring ones:
+- task_description: what phrasing, directness, or output-format specification appears in the highest-scoring configs?
+- system_behavior: what reasoning strategy, focus angle, or decision heuristic appears in the highest-scoring configs?
+- output_constraints: what format rule β strictness of brevity, structure, or exclusions β correlates with higher scores?
+
+Write ONE updated actionable hint (under 50 words) covering the strongest pattern across all three fields.
+Wrap the hint with .
+"""
+
+CROSSOVER_TEMPLATE_COEVO_ENH = """You are an expert in prompt optimization. Design an improved three-field prompt configuration.
+
+Task: {PROBLEM_DESCRIPTION}
+
+[Worse configuration] (score: {WORSE_SCORE})
+task_description: {WORSE_PROMPT_TEXT}
+system_behavior: {WORSE_PROMPT_ROLE}
+output_constraints: {WORSE_PROMPT_CONSTRAINTS}
+
+[Better configuration] (score: {BETTER_SCORE})
+task_description: {BETTER_PROMPT_TEXT}
+system_behavior: {BETTER_PROMPT_ROLE}
+output_constraints: {BETTER_PROMPT_CONSTRAINTS}
+
+[Key insight from comparing these configurations]
+{SHORT_TERM_REFLECTION}
+
+Combine the strongest element from each configuration. You may take any field unchanged from either configuration, or write a new version of a field guided by the insight above.
+Goal: score above {BETTER_SCORE}.
+
+Field rules (strictly enforced):
+- "task_description": WHAT to do and what output format is expected. 1β2 sentences. No reasoning instructions.
+- "system_behavior": HOW to approach the task β reasoning strategy, what to prioritize, specific checks, or default decisions when input is ambiguous. Can start "You are [brief role]" ONLY if immediately followed by a concrete behavioral instruction. 1β2 sentences, 8β25 words. Must NOT repeat task_description content.
+- "output_constraints": OUTPUT FORMAT rules only β length limits, structure, what to include or exclude in the response. Must NOT include reasoning instructions, decision strategies, or content already stated in the other two fields. 1β2 short rules.
+
+Output JSON only:
+{{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}}
+"""
+
+MUTATION_TEMPLATE_COEVO_ENH = """You are an expert in prompt optimization. Generate a targeted mutation of the current best configuration.
+
+Task: {PROBLEM_DESCRIPTION}
+
+[Accumulated insight on what works for this task]
+{LONG_TERM_REFLECTION}
+
+[Current best configuration] (score: {ELITIST_SCORE})
+task_description: {ELITIST_PROMPT_TEXT}
+system_behavior: {ELITIST_PROMPT_ROLE}
+output_constraints: {ELITIST_PROMPT_CONSTRAINTS}
+
+[Cases where the current configuration most often fails]
+Each line shows: input | wrong output the model gave | correct answer.
+{BAD_EXAMPLES}
+
+Before writing, diagnose which field is responsible for these failures:
+- task_description issue: does the instruction fail to convey the right output scope, format, or distinction between cases?
+- system_behavior issue: does the reasoning strategy fail to handle the specific input patterns shown above, or is it biased toward certain classes?
+- output_constraints issue: does the model produce extra text, wrong structure, or wrong format that hurts scoring?
+
+Mutate the field(s) most responsible for the failures. The other fields may stay the same or be improved moderately.
+Goal: score above {ELITIST_SCORE}.
+
+Field rules (strictly enforced):
+- "task_description": WHAT to do and what output format is expected. 1β2 sentences.
+- "system_behavior": HOW to approach the task β reasoning strategy, what to check, default decisions for ambiguous cases. Can start "You are [brief role]" ONLY if immediately followed by a concrete behavioral instruction. 1β2 sentences, 8β25 words. Must NOT repeat task_description content.
+- "output_constraints": OUTPUT FORMAT rules only β length, structure, what to include or exclude in the response. Must NOT include reasoning instructions, decision strategies, or content already stated in other fields. 1β2 short rules.
+
+Output JSON only:
+{{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}}
+"""
+
+PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_ENH = """Generate a concise initial three-field prompt configuration for the following task.
+
+Task: {PROBLEM_DESCRIPTION}
+
+Output a JSON object with exactly three fields:
+- "task_description": What the model should do and how to return the answer (1β2 sentences). Be specific about the output format (e.g., a label, a number, a single sentence, the exact expected structure).
+- "system_behavior": How the model should approach the task β one concrete reasoning principle or focus area (1 sentence, 8β20 words). Must be a behavioral instruction, not just a role title.
+ Good: "Check for ambiguous cases before deciding." / "Focus on the key entity before forming a response."
+ Bad: "You are an expert." β vague, no behavioral instruction.
+- "output_constraints": Output format rules only β what to include or exclude in the response (1β2 short rules, under 15 words total).
+ Must NOT include reasoning instructions or decision strategies β those belong in system_behavior.
+
+Keep all fields brief and generic. These are starting points to be refined by the optimizer.
+Output JSON only.
+"""
diff --git a/coolprompt/utils/prompt_templates/reflective_templates_coevo_per_field.py b/coolprompt/utils/prompt_templates/reflective_templates_coevo_per_field.py
new file mode 100644
index 00000000..d3555b36
--- /dev/null
+++ b/coolprompt/utils/prompt_templates/reflective_templates_coevo_per_field.py
@@ -0,0 +1,114 @@
+from coolprompt.utils.prompt_templates.reflective_templates_coevo_enhanced import (
+ PARAPHRASING_TEMPLATE_COEVO_ENH as PARAPHRASING_TEMPLATE_COEVO_PF,
+ PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_ENH as PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_PF,
+)
+
+SHORT_TERM_REFLECTION_TEMPLATE_COEVO_PF = """You are an expert in prompt optimization. Compare two three-field configurations and identify what makes the better one score higher.
+
+Task: {PROBLEM_DESCRIPTION}
+
+[Worse configuration] (score: {WORSE_SCORE})
+task_description: {WORSE_PROMPT_TEXT}
+system_behavior: {WORSE_PROMPT_ROLE}
+output_constraints: {WORSE_PROMPT_CONSTRAINTS}
+
+[Better configuration] (score: {BETTER_SCORE})
+task_description: {BETTER_PROMPT_TEXT}
+system_behavior: {BETTER_PROMPT_ROLE}
+output_constraints: {BETTER_PROMPT_CONSTRAINTS}
+
+Analyze what makes the better configuration score higher. Then write three separate actionable hints, one per field (under 20 words each).
+Wrap each hint in its own tags:
+- task_description hint: wrap with
+- system_behavior hint: wrap with
+- output_constraints hint: wrap with
+"""
+
+LONG_TERM_REFLECTION_TEMPLATE_COEVO_PF = """You are an expert in prompt optimization. Synthesize patterns from the best-performing configurations found so far.
+
+Task: {PROBLEM_DESCRIPTION}
+
+Best configurations found so far (ranked by score, best first):
+{TOP_PROMPTS_HISTORY}
+
+Prior accumulated per-field insights:
+task_description: {PRIOR_TASK_HINT}
+system_behavior: {PRIOR_ROLE_HINT}
+output_constraints: {PRIOR_CONSTRAINTS_HINT}
+
+New per-field observations from recent comparisons:
+{NEW_SHORT_TERM_REFLECTIONS}
+
+Study the top configurations above and update each per-field insight. Write one updated actionable hint per field (under 30 words each) covering the strongest pattern.
+Wrap each hint in its own tags:
+- task_description hint: wrap with
+- system_behavior hint: wrap with
+- output_constraints hint: wrap with
+"""
+
+CROSSOVER_TEMPLATE_COEVO_PF = """You are an expert in prompt optimization. Design an improved three-field prompt configuration.
+
+Task: {PROBLEM_DESCRIPTION}
+
+[Worse configuration] (score: {WORSE_SCORE})
+task_description: {WORSE_PROMPT_TEXT}
+system_behavior: {WORSE_PROMPT_ROLE}
+output_constraints: {WORSE_PROMPT_CONSTRAINTS}
+
+[Better configuration] (score: {BETTER_SCORE})
+task_description: {BETTER_PROMPT_TEXT}
+system_behavior: {BETTER_PROMPT_ROLE}
+output_constraints: {BETTER_PROMPT_CONSTRAINTS}
+
+[Per-field insights from comparing these configurations]
+task_description: {TASK_HINT}
+system_behavior: {ROLE_HINT}
+output_constraints: {CONSTRAINTS_HINT}
+
+Combine the strongest element from each configuration guided by the field-specific insights above.
+You may take any field unchanged from either configuration, or write a new version guided by its insight.
+Goal: score above {BETTER_SCORE}.
+
+Field rules (strictly enforced):
+- "task_description": WHAT to do and what output format is expected. 1β2 sentences. No reasoning instructions.
+- "system_behavior": HOW to approach the task β reasoning strategy, what to prioritize, specific checks, or default decisions when input is ambiguous. Can start "You are [brief role]" ONLY if immediately followed by a concrete behavioral instruction. 1β2 sentences, 8β25 words. Must NOT repeat task_description content.
+- "output_constraints": OUTPUT FORMAT rules only β length limits, structure, what to include or exclude in the response. Must NOT include reasoning instructions, decision strategies, or content already stated in the other two fields. 1β2 short rules.
+
+Output JSON only:
+{{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}}
+"""
+
+MUTATION_TEMPLATE_COEVO_PF = """You are an expert in prompt optimization. Generate a targeted mutation of the current best configuration.
+
+Task: {PROBLEM_DESCRIPTION}
+
+[Accumulated per-field insights on what works for this task]
+task_description: {TASK_HINT}
+system_behavior: {ROLE_HINT}
+output_constraints: {CONSTRAINTS_HINT}
+
+[Current best configuration] (score: {ELITIST_SCORE})
+task_description: {ELITIST_PROMPT_TEXT}
+system_behavior: {ELITIST_PROMPT_ROLE}
+output_constraints: {ELITIST_PROMPT_CONSTRAINTS}
+
+[Cases where the current configuration most often fails]
+Each line shows: input | wrong output the model gave | correct answer.
+{BAD_EXAMPLES}
+
+Before writing, diagnose which field is responsible for these failures:
+- task_description: does the instruction fail to convey the right output scope, format, or distinction between cases?
+- system_behavior: does the reasoning strategy fail to handle the specific input patterns shown above, or is it biased toward certain classes?
+- output_constraints: does the model produce extra text, wrong structure, or wrong format that hurts scoring?
+
+Mutate the field(s) most responsible for the failures, guided by the per-field insights above.
+Goal: score above {ELITIST_SCORE}.
+
+Field rules (strictly enforced):
+- "task_description": WHAT to do and what output format is expected. 1β2 sentences.
+- "system_behavior": HOW to approach the task β reasoning strategy, what to check, default decisions for ambiguous cases. Can start "You are [brief role]" ONLY if immediately followed by a concrete behavioral instruction. 1β2 sentences, 8β25 words. Must NOT repeat task_description content.
+- "output_constraints": OUTPUT FORMAT rules only β length, structure, what to include or exclude in the response. Must NOT include reasoning instructions, decision strategies, or content already stated in other fields. 1β2 short rules.
+
+Output JSON only:
+{{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}}
+"""
diff --git a/coolprompt/utils/prompt_templates/reflective_templates_coevolution.py b/coolprompt/utils/prompt_templates/reflective_templates_coevolution.py
new file mode 100644
index 00000000..c9ba550b
--- /dev/null
+++ b/coolprompt/utils/prompt_templates/reflective_templates_coevolution.py
@@ -0,0 +1,351 @@
+REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_BASE = """You are an expert in prompt optimization. Your task is to give hints to design better prompt configurations.
+
+Below are two prompt configurations for {PROBLEM_DESCRIPTION}.
+Each configuration has two components:
+- system_behavior: behavioral instructions defining HOW the AI should reason, verify, and process information (NOT a persona or job title)
+- task_description: the specific task instruction defining WHAT the AI should do
+
+The second configuration performs better than the first one.
+[Worse configuration]
+System Behavior: {WORSE_PROMPT_ROLE}
+Task Description: {WORSE_PROMPT_TEXT}
+[Better configuration]
+System Behavior: {BETTER_PROMPT_ROLE}
+Task Description: {BETTER_PROMPT_TEXT}
+Analyze differences in both system_behavior and task_description separately.
+Consider WHY the better configuration works better. Focus on actionable changes.
+Respond with one concise hint (less than 30 words) covering what to change in system_behavior and what to change in task_description.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_BASE = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks.
+Your response outputs a JSON object with two fields: "system_behavior" and "task_description".
+
+Write a new prompt configuration for the task: {PROBLEM_DESCRIPTION}.
+
+[Worse configuration]
+System Behavior: {WORSE_PROMPT_ROLE}
+Task Description: {WORSE_PROMPT_TEXT}
+[Better configuration]
+System Behavior: {BETTER_PROMPT_ROLE}
+Task Description: {BETTER_PROMPT_TEXT}
+[Reflection]
+{SHORT_TERM_REFLECTION}
+[Improved configuration]
+Combine the strongest aspects of both configurations according to the reflection.
+You may take the system_behavior approach from one configuration and the task_description approach from the other.
+
+Rules:
+- "system_behavior" MUST describe specific ACTIONS and REASONING STEPS β not identity or expertise.
+ Do NOT write: "You are an expert in X", "A specialist in Y", "Domain professional"
+ DO write: "Before answering, verify X. Check for Y. If Z, then..."
+- "system_behavior" MUST be a complete instruction of at least 8 words.
+- "task_description" MUST be a clear, actionable task instruction.
+- system_behavior and task_description must complement each other without duplicating instructions.
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_BASE = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks.
+Your response outputs a JSON object with two fields: "system_behavior" and "task_description".
+
+Write a mutated prompt configuration for {PROBLEM_DESCRIPTION}.
+[Prior reflection]
+{LONG_TERM_REFLECTION}
+[Current elitist configuration]
+System Behavior: {ELITIST_PROMPT_ROLE}
+Task Description: {ELITIST_PROMPT_TEXT}
+[Mutated configuration]
+IMPORTANT for system_behavior: Aggressively reimagine it. Create a fundamentally different behavioral specification.
+Do NOT describe identity (who you are). Describe BEHAVIOR (what to do, what to check, how to reason).
+system_behavior MUST be a complete instruction of at least 8 words β NOT a persona or job title.
+Bad examples: "Topic Analyst", "Data Specialist", "You are an expert in X"
+Good examples: "Before responding, verify the key facts in the input. Check for consistency and relevance.", "Identify the core information first, then formulate a concise and accurate response."
+The task_description may be changed moderately, applying the accumulated reflection.
+system_behavior and task_description must not repeat the same instructions.
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO ="""You are an expert in prompt optimization. Your task is to give hints to design better prompt configurations.
+
+Below are two prompt configurations for {PROBLEM_DESCRIPTION}.
+Each configuration has two components:
+- task_description: WHAT the model should do and how to format the answer. Covers the task itself and output requirements.
+- system_behavior: HOW the model should approach the task β reasoning strategy, key things to check, special considerations. Can start with a brief role ("You are X") only if immediately followed by a concrete behavioral instruction.
+
+The second configuration performs better than the first one.
+[Worse configuration] (score: {WORSE_SCORE})
+System Behavior: {WORSE_PROMPT_ROLE}
+Task Description: {WORSE_PROMPT_TEXT}
+[Better configuration] (score: {BETTER_SCORE})
+System Behavior: {BETTER_PROMPT_ROLE}
+Task Description: {BETTER_PROMPT_TEXT}
+Analyze differences in both components separately.
+Consider WHY the better configuration scored higher on this specific task.
+Respond with one concise hint (less than 30 words) about what makes the better configuration work.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO = """You are an expert in prompt optimization. Your task is to give hints to design better prompt configurations.
+
+Task: {PROBLEM_DESCRIPTION}
+
+Best-performing configurations found so far (ranked by score, best first):
+{TOP_PROMPTS_HISTORY}
+
+Study these top configurations: what patterns in system_behavior and task_description do the higher-scoring ones share?
+
+Prior accumulated insight:
+{PRIOR_LONG_TERM_REFLECTION}
+
+New observations from recent comparisons:
+{NEW_SHORT_TERM_REFLECTIONS}
+
+Write one updated actionable hint (less than 50 words) about what makes a configuration score highest on this specific task.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks.
+Your response outputs a JSON object with two fields: "system_behavior" and "task_description".
+
+Write a new prompt configuration for the task: {PROBLEM_DESCRIPTION}.
+
+[Worse configuration] (score: {WORSE_SCORE})
+System Behavior: {WORSE_PROMPT_ROLE}
+Task Description: {WORSE_PROMPT_TEXT}
+[Better configuration] (score: {BETTER_SCORE})
+System Behavior: {BETTER_PROMPT_ROLE}
+Task Description: {BETTER_PROMPT_TEXT}
+[Reflection]
+{SHORT_TERM_REFLECTION}
+[Improved configuration]
+Combine the strongest aspects of both configurations according to the reflection.
+Your goal is to score HIGHER than {BETTER_SCORE}.
+
+Field rules:
+- "task_description": WHAT to do and how to format the answer. Clear and actionable. 1-2 sentences.
+- "system_behavior": HOW to approach the task β reasoning strategy, what to prioritize, specific checks.
+ You CAN start with "You are [brief role]" if you immediately follow it with a concrete behavioral instruction.
+ Example: "You are a careful analyst. Focus on the most relevant detail and verify it matches the expected format."
+ NOT enough: "You are an expert." β must say what to DO or CHECK.
+ 1-2 sentences, 8-25 words.
+- The two fields must cover different aspects β task_description covers the task itself, system_behavior covers the approach. Do not repeat the same instruction in both.
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks.
+Your response outputs a JSON object with two fields: "system_behavior" and "task_description".
+
+Task: {PROBLEM_DESCRIPTION}
+[Accumulated insight on what works for this task]
+{LONG_TERM_REFLECTION}
+
+[Current best configuration] (score: {ELITIST_SCORE})
+System Behavior: {ELITIST_PROMPT_ROLE}
+Task Description: {ELITIST_PROMPT_TEXT}
+
+[Examples where the current configuration most often fails]
+Each example shows the input, the wrong answer the model gave, and the correct answer.
+{BAD_EXAMPLES}
+
+Analyze these failure cases before writing:
+- Does the failure come from the task instruction (task_description) or from the behavioral approach (system_behavior)?
+- What specific reasoning step or focus is missing that would fix these cases?
+- What change to either field would prevent these specific errors?
+
+Write a mutated configuration that directly addresses the identified failure pattern and aims to score above {ELITIST_SCORE}.
+
+Field rules:
+- "task_description": WHAT to do and how to format the answer. 1-2 sentences.
+- "system_behavior": HOW to approach the task β reasoning strategy, what to check, special considerations.
+ You CAN start with "You are [brief role]" if you immediately follow it with a concrete behavioral instruction.
+ Example: "You are a careful analyst. Focus on the most relevant detail before committing to an answer."
+ NOT enough: "You are an expert." β must say what to DO or CHECK.
+ 1-2 sentences, 8-25 words.
+- The two fields must cover different aspects. Do not repeat the same instruction in both.
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO = """Create diverse variations of the given prompt configuration.
+System Behavior: {ROLE}
+Task Description: {PROMPT}
+
+Create {NUM_PROMPTS} variations with maximum diversity.
+Each variation must have a meaningfully DIFFERENT system_behavior that takes a unique behavioral approach.
+
+Field rules:
+- "task_description": WHAT to do and how to format the answer. Vary wording while preserving task intent.
+- "system_behavior": HOW to approach the task β reasoning strategy, what to prioritize, specific checks. 1-2 sentences.
+ You CAN start with "You are [brief role]" if you immediately follow it with a concrete behavioral instruction.
+ Examples: "You are a precise reader. Focus on the key entity before formulating a response.",
+ "Before responding, identify the main requirement and verify your answer matches it.",
+ "You are a methodical solver. Break the input into parts and handle each systematically."
+ NOT enough: "You are an expert." β must state what to DO or CHECK.
+- The two fields must cover different aspects. Do not repeat the same instruction in both.
+
+Output them in JSON structure below:
+{{
+ "prompts": [
+ {{"system_behavior": "behavior 1", "task_description": "task 1"}},
+ {{"system_behavior": "behavior 2", "task_description": "task 2"}},
+ ...
+ {{"system_behavior": "behavior {NUM_PROMPTS}", "task_description": "task {NUM_PROMPTS}"}},
+ ]
+}}
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO = """Generate a simple initial prompt configuration for the task: {PROBLEM_DESCRIPTION}.
+
+Output a JSON object with exactly two fields:
+- "system_behavior": A SHORT behavioral hint (1 sentence, max 15 words) about HOW to approach the task.
+ Write a simple practical instruction, NOT a role or identity.
+ Bad: "You are an expert in X", "Data Analyst"
+ Good: "Think step by step before answering.", "Check your reasoning carefully."
+- "task_description": A SHORT task instruction (1 sentence, max 15 words) about WHAT to do.
+
+IMPORTANT: Keep BOTH fields brief and generic. These are starting points that will be refined later.
+Do NOT write elaborate multi-step strategies. Do NOT include specific examples or edge cases.
+Output JSON only.
+"""
+
+REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_3F = """You are an expert in prompt optimization. Your task is to give hints to design better prompt configurations.
+
+Below are two prompt configurations for {PROBLEM_DESCRIPTION}.
+Each configuration has three components:
+- task_description: WHAT the model should do and how to format the answer. Covers the task itself and output requirements.
+- system_behavior: HOW the model should approach the task β reasoning strategy, what to check, special considerations. Can start with a brief role ("You are X") only if immediately followed by a concrete behavioral instruction.
+- output_constraints: FORMAT and STYLE rules only β length limits, structure, tone, what to omit. Examples: "One sentence only.", "No extra context.", "Use subject-verb-object order."
+
+The second configuration performs better than the first one.
+[Worse configuration] (score: {WORSE_SCORE})
+System Behavior: {WORSE_PROMPT_ROLE}
+Task Description: {WORSE_PROMPT_TEXT}
+Output Constraints: {WORSE_PROMPT_CONSTRAINTS}
+[Better configuration] (score: {BETTER_SCORE})
+System Behavior: {BETTER_PROMPT_ROLE}
+Task Description: {BETTER_PROMPT_TEXT}
+Output Constraints: {BETTER_PROMPT_CONSTRAINTS}
+Analyze differences in all three components separately.
+Consider WHY the better configuration scored higher on this specific task.
+Respond with one concise hint (less than 30 words) about what makes the better configuration work.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO_3F = """You are an expert in prompt optimization. Your task is to give hints to design better prompt configurations.
+
+Task: {PROBLEM_DESCRIPTION}
+
+Best-performing configurations found so far (ranked by score, best first):
+{TOP_PROMPTS_HISTORY}
+
+Study these top configurations: what patterns in system_behavior, task_description, and output_constraints do the higher-scoring ones share?
+
+Prior accumulated insight:
+{PRIOR_LONG_TERM_REFLECTION}
+
+New observations from recent comparisons:
+{NEW_SHORT_TERM_REFLECTIONS}
+
+Write one updated actionable hint (less than 50 words) about what makes a configuration score highest on this task.
+Cover three aspects: what system_behavior patterns work best, what task_description patterns work best, and what output_constraints are most effective.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_3F = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks.
+Your response outputs a JSON object with three fields: "system_behavior", "task_description", and "output_constraints".
+
+Write a new prompt configuration for the task: {PROBLEM_DESCRIPTION}.
+
+[Worse configuration] (score: {WORSE_SCORE})
+System Behavior: {WORSE_PROMPT_ROLE}
+Task Description: {WORSE_PROMPT_TEXT}
+Output Constraints: {WORSE_PROMPT_CONSTRAINTS}
+[Better configuration] (score: {BETTER_SCORE})
+System Behavior: {BETTER_PROMPT_ROLE}
+Task Description: {BETTER_PROMPT_TEXT}
+Output Constraints: {BETTER_PROMPT_CONSTRAINTS}
+[Reflection]
+{SHORT_TERM_REFLECTION}
+[Improved configuration]
+Combine the strongest aspects of both configurations according to the reflection.
+Your goal is to score HIGHER than {BETTER_SCORE}.
+
+Field rules:
+- "task_description": WHAT to do and how to format the answer. 1-2 sentences. Clear and actionable.
+- "system_behavior": HOW to approach the task β reasoning strategy, what to prioritize, specific checks.
+ You CAN start with "You are [brief role]" if immediately followed by a concrete behavioral instruction.
+ Example: "You are a careful analyst. Focus on the most relevant detail and verify it matches the expected format."
+ NOT enough: "You are an expert." β must say what to DO or CHECK. 1-2 sentences, 8-25 words.
+- "output_constraints": FORMAT and STYLE rules only β length, structure, what to omit. 1-2 short rules.
+ Example: "One sentence only. No extra context beyond the main point."
+- The three fields must cover different aspects β no repeated instructions across fields.
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_3F = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks.
+Your response outputs a JSON object with three fields: "system_behavior", "task_description", and "output_constraints".
+
+Task: {PROBLEM_DESCRIPTION}
+[Accumulated insight on what works for this task]
+{LONG_TERM_REFLECTION}
+
+[Current best configuration] (score: {ELITIST_SCORE})
+System Behavior: {ELITIST_PROMPT_ROLE}
+Task Description: {ELITIST_PROMPT_TEXT}
+Output Constraints: {ELITIST_PROMPT_CONSTRAINTS}
+
+[Examples where the current configuration most often fails]
+Each example shows the input, the wrong answer the model gave, and the correct answer.
+{BAD_EXAMPLES}
+
+Analyze these failure cases before writing:
+- Does the failure come from task_description (wrong instruction), system_behavior (wrong approach), or output_constraints (wrong format rule)?
+- What specific change to which field would prevent these errors?
+
+Write a mutated configuration that directly addresses the identified failure pattern and aims to score above {ELITIST_SCORE}.
+
+Field rules:
+- "task_description": WHAT to do and how to format the answer. 1-2 sentences.
+- "system_behavior": HOW to approach the task β reasoning strategy, what to check, special considerations.
+ You CAN start with "You are [brief role]" if immediately followed by a concrete behavioral instruction.
+ Example: "You are a careful analyst. Focus on the most relevant detail before committing to an answer."
+ NOT enough: "You are an expert." β must say what to DO or CHECK. 1-2 sentences, 8-25 words.
+- "output_constraints": FORMAT and STYLE rules only β length limits, structure, what to include or omit.
+ Example: "One sentence only. No additional context. Start directly with the answer."
+- The three fields must cover different aspects. No repeated instructions across fields.
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO_3F = """Create diverse variations of the given prompt configuration.
+System Behavior: {ROLE}
+Task Description: {PROMPT}
+Output Constraints: {CONSTRAINTS}
+Create {NUM_PROMPTS} variations with maximum diversity.
+Each variation must have meaningfully DIFFERENT system_behavior, task_description, and output_constraints.
+- "system_behavior": Start with a role identity, then describe reasoning steps.
+- "task_description": Vary wording while preserving task intent.
+- "output_constraints": Vary format, length, tone, or quality rules.
+Output them in JSON structure below:
+{{
+ "prompts": [
+ {{"system_behavior": "New behavior 1", "task_description": "New task 1", "output_constraints": "New constraints 1"}},
+ {{"system_behavior": "New behavior 2", "task_description": "New task 2", "output_constraints": "New constraints 2"}},
+ ...
+ {{"system_behavior": "New behavior {NUM_PROMPTS}", "task_description": "New task {NUM_PROMPTS}", "output_constraints": "New constraints {NUM_PROMPTS}"}},
+ ]
+}}
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_3F = """Generate a simple initial prompt configuration for the task: {PROBLEM_DESCRIPTION}.
+
+Output a JSON object with exactly three fields:
+- "system_behavior": A brief role identity + practical reasoning hint (max 20 words).
+ Example: "You are a logical analyst. Think step by step before answering."
+- "task_description": A SHORT task instruction (1 sentence, max 15 words).
+- "output_constraints": Brief rules about output format or style (max 15 words).
+ Example: "Be concise. Follow the requested format strictly."
+
+IMPORTANT: Keep ALL fields brief and generic. These are starting points that will be refined later.
+Output JSON only.
+"""
diff --git a/coolprompt/utils/prompt_templates/reflective_templates_factorized.py b/coolprompt/utils/prompt_templates/reflective_templates_factorized.py
new file mode 100644
index 00000000..7423da1f
--- /dev/null
+++ b/coolprompt/utils/prompt_templates/reflective_templates_factorized.py
@@ -0,0 +1,305 @@
+
+REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_ROLE_ONLY = """Create {NUM_PROMPTS} diverse system_behavior variants for a fixed task instruction.
+
+Task: {PROBLEM_DESCRIPTION}
+Task instruction (frozen β do not change): {PROMPT}
+
+Current system_behavior:
+{ROLE}
+
+Rules:
+- Each system_behavior must be 1 sentence, between 8 and 25 words.
+- You CAN start with "You are X" if you immediately follow it with a concrete behavior or focus.
+ It is NOT enough to only name a role β you must say what the model should DO or NOTICE.
+- Vary the framing: some variants should name a focus area, some a cognitive strategy, some a domain angle.
+- The variants should differ meaningfully from each other.
+- Do NOT copy examples from below β they are for illustration only, from unrelated domains.
+ Example (translation task): "You are a careful translator. Preserve the original register and avoid literal word-for-word rendering."
+ Example (legal task): "Identify the main obligation being described before formulating an answer."
+ Example (coding task): "You are a code reviewer. Flag potential edge cases as well as the obvious issue."
+
+Output JSON:
+{{
+ "prompts": [
+ {{"system_behavior": "variant 1"}},
+ {{"system_behavior": "variant 2"}},
+ ...
+ {{"system_behavior": "variant {NUM_PROMPTS}"}}
+ ]
+}}
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_ROLE_ONLY = """You are an expert in prompt optimization. Give a hint for writing better system_behavior instructions.
+
+Task: {PROBLEM_DESCRIPTION}
+Task instruction (fixed): {FROZEN_PROMPT_TEXT}
+
+Two system_behavior instructions were tested.
+[Worse system_behavior] (score: {WORSE_SCORE})
+{WORSE_PROMPT_ROLE}
+[Better system_behavior] (score: {BETTER_SCORE})
+{BETTER_PROMPT_ROLE}
+
+Why does the better framing lead to higher scores on this specific task?
+Respond with one hint in less than 20 words. Focus on what the better framing emphasizes.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_ROLE_ONLY = """You are an expert in prompt optimization. Synthesize hints for writing better system_behavior instructions.
+
+Task: {PROBLEM_DESCRIPTION}
+
+Best-performing system_behavior instructions found so far (ranked by score, best first):
+{TOP_PROMPTS_HISTORY}
+
+Study these top system_behavior instructions: what framing, cognitive strategy, or focus angle do the higher-scoring ones use that lower-scoring ones lack?
+
+Prior accumulated insight:
+{PRIOR_LONG_TERM_REFLECTION}
+
+New observations from recent comparisons:
+{NEW_SHORT_TERM_REFLECTIONS}
+
+Write one updated actionable hint (less than 40 words) about what framing in system_behavior scores highest on this specific task.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_ROLE_ONLY = """You are an expert in prompt optimization. Design a better system_behavior instruction.
+
+Task: {PROBLEM_DESCRIPTION}
+Task instruction (frozen): {FROZEN_PROMPT_TEXT}
+
+[Worse system_behavior]
+{WORSE_PROMPT_ROLE}
+[Better system_behavior]
+{BETTER_PROMPT_ROLE}
+[Reflection]
+{SHORT_TERM_REFLECTION}
+
+Write a new system_behavior that takes the best aspect of both.
+Rules:
+- 1-2 sentences, 8-25 words total.
+- You CAN start with "You are X" if you follow it with a concrete behavior.
+ Pure labels without behavior (e.g., "You are an expert.") score poorly.
+- Do NOT use domain-specific terms from the examples below β they are from unrelated tasks:
+ "Identify the key entity being described before forming a response."
+ "Weigh the broader context before committing to a specific category."
+ "You are a precise reader. Prioritize the most prominent feature over peripheral details."
+Output JSON: {{"system_behavior": ""}}
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_MUTATION_TEMPLATE_ROLE_ONLY = """You are an expert in prompt optimization. Generate a new system_behavior instruction.
+
+Task: {PROBLEM_DESCRIPTION}
+Task instruction (frozen): {FROZEN_PROMPT_TEXT}
+
+[Accumulated insight on what framing works for this task]
+{LONG_TERM_REFLECTION}
+
+[Current best system_behavior] (score: {ELITIST_SCORE})
+{ELITIST_PROMPT_ROLE}
+
+[Examples where the current system_behavior most often fails]
+Each example shows the input, the wrong answer the model gave, and the correct answer.
+{BAD_EXAMPLES}
+
+Analyze these failure cases before writing:
+- What type of inputs or edge cases does the current system_behavior fail to handle?
+- Is there a pattern in the errors (e.g., the model misjudges a specific input type)?
+- What cognitive strategy or focus angle could help the model handle these cases correctly?
+
+Write a new system_behavior that directly addresses the identified failure pattern.
+Rules:
+- 1-2 sentences, 8-25 words total.
+- You CAN start with "You are X" if you follow it with a concrete behavior or focus angle.
+ Pure persona labels alone (e.g., "You are an expert.") are not useful.
+- Short roles generalize better β avoid multi-clause chains.
+- Do NOT use domain-specific terms from the examples below β they are from unrelated tasks:
+ "Identify the key entity being described before forming a response."
+ "Weigh the broader context before committing to a specific category."
+ "You are a precise reader. Prioritize the most prominent feature over peripheral details."
+Output JSON: {{"system_behavior": ""}}
+Output JSON data only.
+"""
+
+
+REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_CONSTRAINTS_ONLY = """Create diverse output_constraints variants for a fixed prompt configuration.
+
+Task: {PROBLEM_DESCRIPTION}
+Task Description (frozen): {PROMPT}
+System Behavior (frozen): {ROLE}
+
+Current output_constraints to vary from:
+{CONSTRAINTS}
+
+Create {NUM_PROMPTS} output_constraints variants with maximum diversity.
+output_constraints must only contain OUTPUT FORMAT rules: what the response must or must not contain, length limits, and structural requirements.
+Do NOT include in output_constraints:
+- Reasoning instructions ("think step by step", "analyze X before deciding")
+- Classification strategies ("use X as the default", "prefer Y when ambiguous") β those belong in system_behavior.
+- Instructions that repeat what the task_description or system_behavior already says.
+Do NOT repeat instructions already present in the task_description or system_behavior.
+Examples of good output_constraints (output format rules only):
+- "Return only the final answer. Do not include any explanation or preamble."
+- "Write no more than one sentence. Use plain language."
+- "Output only the requested value with no surrounding text."
+
+Output JSON:
+{{
+ "prompts": [
+ {{"output_constraints": "variant 1"}},
+ {{"output_constraints": "variant 2"}},
+ ...
+ {{"output_constraints": "variant {NUM_PROMPTS}"}}
+ ]
+}}
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY = """You are an expert in prompt optimization. Your task is to give hints for designing better output constraints.
+
+Task: {PROBLEM_DESCRIPTION}
+Task Description (fixed): {FROZEN_PROMPT_TEXT}
+System Behavior (fixed): {FROZEN_PROMPT_ROLE}
+
+Two output_constraints configurations were tested. The second performs better.
+[Worse output_constraints] (score: {WORSE_SCORE})
+{WORSE_PROMPT_CONSTRAINTS}
+[Better output_constraints] (score: {BETTER_SCORE})
+{BETTER_PROMPT_CONSTRAINTS}
+
+Analyze WHY the better output FORMAT rule leads to higher scores on this task.
+Consider: does stricter output brevity help? Does removing explanation noise improve parsing? Does a cleaner response structure match the evaluation metric better?
+Note: output_constraints should cover FORMAT only (length, structure, what to include/exclude in the response). Do NOT suggest classification strategies or reasoning instructions β those belong in system_behavior.
+Respond with one concise hint (less than 30 words).
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY = """You are an expert in prompt optimization. Your task is to give hints for designing better output constraints.
+
+Task: {PROBLEM_DESCRIPTION}
+Task Description (fixed): {FROZEN_PROMPT_TEXT}
+System Behavior (fixed): {FROZEN_PROMPT_ROLE}
+
+Best-performing output_constraints found so far (ranked by score, best first):
+{TOP_PROMPTS_HISTORY}
+
+Study these top constraints: what format rules (brevity, structure, what to exclude) do the higher-scoring ones enforce that lower-scoring ones do not?
+
+Prior accumulated insight:
+{PRIOR_LONG_TERM_REFLECTION}
+
+New observations from recent comparisons:
+{NEW_SHORT_TERM_REFLECTIONS}
+
+Write one updated actionable hint (less than 50 words) summarizing what output constraint patterns score highest on this specific task.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_CONSTRAINTS_ONLY = """You are an expert in prompt optimization. Your task is to design better output constraints.
+
+Task: {PROBLEM_DESCRIPTION}
+Task Description (frozen): {FROZEN_PROMPT_TEXT}
+System Behavior (frozen): {FROZEN_PROMPT_ROLE}
+
+[Worse output_constraints] (score: {WORSE_SCORE})
+{WORSE_PROMPT_CONSTRAINTS}
+[Better output_constraints] (score: {BETTER_SCORE})
+{BETTER_PROMPT_CONSTRAINTS}
+[Reflection]
+{SHORT_TERM_REFLECTION}
+
+Write new output_constraints that combine the strongest aspects of both, targeting a score above {BETTER_SCORE}.
+Rules:
+- Must only cover OUTPUT FORMAT: what the response must/must not contain, length, structure.
+- Must NOT contain reasoning instructions, chain-of-thought requirements, or classification strategies (which label to choose when uncertain) β those belong in system_behavior.
+- Must NOT repeat instructions already in task_description or system_behavior.
+- Keep it concise (1-2 sentences).
+Output JSON: {{"output_constraints": ""}}
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_MUTATION_TEMPLATE_CONSTRAINTS_ONLY = """You are an expert in prompt optimization. Your task is to design new output constraints.
+
+Task: {PROBLEM_DESCRIPTION}
+Task Description (frozen): {FROZEN_PROMPT_TEXT}
+System Behavior (frozen): {FROZEN_PROMPT_ROLE}
+
+[Accumulated insight on what constraint patterns work best]
+{LONG_TERM_REFLECTION}
+
+[Current best output_constraints] (score: {ELITIST_SCORE})
+{ELITIST_PROMPT_CONSTRAINTS}
+
+[Examples where the current constraints most often fail]
+Each example shows the input, the wrong answer the model gave, and the correct answer.
+{BAD_EXAMPLES}
+
+Analyze these failure cases before writing:
+- Does the model output contain extra text, explanation, or preamble that hurts evaluation?
+- Is the response format wrong (e.g., wrong delimiter, extra whitespace, wrong casing)?
+- Would stricter length or structure rules prevent these specific errors?
+
+Generate output_constraints that directly address the identified format failures and target a score above {ELITIST_SCORE}.
+Rules:
+- Must only cover OUTPUT FORMAT: what the response must/must not contain, length, structure.
+- Must NOT contain reasoning instructions, chain-of-thought requirements, or classification strategies β those belong in system_behavior.
+- Must NOT repeat instructions already in task_description or system_behavior.
+- Try varying: length limits ("only the digit", "one sentence max"), format rules ("no preamble", "no explanation"), structural requirements.
+- Keep it concise (1-2 sentences).
+Output JSON: {{"output_constraints": ""}}
+Output JSON data only.
+"""
+
+
+
+DEDUP_ROLE_TEMPLATE = """You are a prompt engineer reviewing a multi-field prompt for redundancy.
+
+The task_description field is fixed:
+TASK_DESCRIPTION: {TASK}
+
+Review this system_behavior:
+SYSTEM_BEHAVIOR: {ROLE}
+
+Remove from SYSTEM_BEHAVIOR any content already covered by TASK_DESCRIPTION.
+Specifically remove:
+- Any restatement of what the task is β already in task_description
+- Any label or value definitions already listed in task_description
+- Any output format rules (e.g. "return only X") already in task_description
+Keep only what is UNIQUE to system_behavior: cognitive strategy, reasoning approach, focus angle, default decisions when uncertain, persona framing.
+If nothing unique remains, return an empty string.
+
+Example of what to remove (translation task, for illustration only):
+ task_description: "Translate the text from English to French."
+ system_behavior: "You are a translator. Translate English text to French. Preserve tone."
+ β Remove "Translate English text to French" (already in task). Keep "Preserve tone."
+
+Return JSON only: {{"system_behavior": ""}}
+Output JSON data only."""
+
+DEDUP_CONSTRAINTS_TEMPLATE = """You are a prompt engineer reviewing a multi-field prompt for redundancy.
+
+The task_description and system_behavior fields are fixed:
+TASK_DESCRIPTION: {TASK}
+SYSTEM_BEHAVIOR: {ROLE}
+
+Review these output_constraints:
+OUTPUT_CONSTRAINTS: {CONSTRAINTS}
+
+Remove from OUTPUT_CONSTRAINTS any content already covered by TASK_DESCRIPTION or SYSTEM_BEHAVIOR.
+Specifically remove:
+- Any label or value definitions already in task_description
+- Any output format rules already specified in task_description
+- Any reasoning instructions or decision strategies already in system_behavior
+Keep only UNIQUE format rules: response length limits, structural requirements, style restrictions not already stated elsewhere.
+If nothing unique remains, return an empty string.
+
+Example of what to remove (legal task, for illustration only):
+ task_description: "Identify the main obligation. Return a single sentence."
+ output_constraints: "Return a single sentence stating the main obligation. Be concise."
+ β Remove "Return a single sentence" (already in task). Keep "Be concise" only if it adds something new.
+
+Return JSON only: {{"output_constraints": ""}}
+Output JSON data only."""
diff --git a/coolprompt/utils/prompt_templates/reflective_templates_fixed_role.py b/coolprompt/utils/prompt_templates/reflective_templates_fixed_role.py
new file mode 100644
index 00000000..a297cb67
--- /dev/null
+++ b/coolprompt/utils/prompt_templates/reflective_templates_fixed_role.py
@@ -0,0 +1,89 @@
+REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_FIXED_ROLE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts.
+
+Below are two prompt configurations for {PROBLEM_DESCRIPTION}.
+You are provided with two prompt versions below, where the second version performs better than the first one.
+
+The System Role is FIXED and is the same for both prompts:
+Role: {BETTER_PROMPT_ROLE}
+
+[Worse prompt text]
+Prompt: {WORSE_PROMPT_TEXT}
+[Better prompt text]
+Prompt: {BETTER_PROMPT_TEXT}
+
+You respond only with one small hint for designing better prompts TEXT, based on the two prompt versions and fixed role, using less than 20 words.
+I want you to generate only one new hint for the prompt text itself. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_FIXED_ROLE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts.
+
+Below is your prior longβterm reflection on designing prompts for {PROBLEM_DESCRIPTION}.
+{PRIOR_LONG_TERM_REFLECTION}
+
+Below are some newly gained insights.
+{NEW_SHORT_TERM_REFLECTIONS}
+
+Write the constructive hint for designing better prompt TEXTS, based on prior reflections and new insights and using less than 50 words.
+The System Role is FIXED, so focus only on optimizing the user prompt text.
+I want you to generate only one new constructive hint. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_FIXED_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems.
+Your response outputs a JSON object with one field: "prompt".
+
+The System Role is FIXED and cannot be changed:
+Role: {BETTER_PROMPT_ROLE}
+
+Write a new prompt text for the task: {PROBLEM_DESCRIPTION}.
+
+[Worse prompt text]
+Prompt: {WORSE_PROMPT_TEXT}
+[Better prompt text]
+Prompt: {BETTER_PROMPT_TEXT}
+[Reflection]
+{SHORT_TERM_REFLECTION}
+[Improved prompt configuration]
+Please write an improved prompt text, according to the reflection, optimized for the fixed role above.
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_MUTATION_TEMPLATE_FIXED_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems.
+Your response outputs a JSON object with one field: "prompt".
+
+The System Role is FIXED and cannot be changed:
+Role: {ELITIST_PROMPT_ROLE}
+
+Write a mutated prompt text for {PROBLEM_DESCRIPTION}.
+[Prior reflection]
+{LONG_TERM_REFLECTION}
+[Current elitist prompt text]
+Prompt: {ELITIST_PROMPT_TEXT}
+[Mutated prompt configuration]
+Please write a mutated prompt text, according to the reflection, optimized for the fixed role above.
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_FIXED_ROLE = """Paraphrase the given prompt text keeping its initial meaning. The role is fixed.
+
+Fixed Role: {ROLE}
+Prompt: {PROMPT}
+
+Create {NUM_PROMPTS} new variations of this prompt text (optimized for the fixed role) and output them in JSON structure below:
+{{
+ "prompts": [
+ "New prompt 1",
+ "New prompt 2",
+ ...
+ "New prompt {NUM_PROMPTS}"
+ ]
+}}
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_FIXED_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems.
+Write a prompt text that will effectively solve the task: {PROBLEM_DESCRIPTION}.
+The System Role is FIXED and provided separately. Focus only on the prompt text.
+Output a JSON object with one field: "prompt".
+"""
diff --git a/coolprompt/utils/prompt_templates/reflective_templates_no_role.py b/coolprompt/utils/prompt_templates/reflective_templates_no_role.py
new file mode 100644
index 00000000..e8ba276c
--- /dev/null
+++ b/coolprompt/utils/prompt_templates/reflective_templates_no_role.py
@@ -0,0 +1,76 @@
+REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_NO_ROLE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts.
+
+Below are two prompts for {PROBLEM_DESCRIPTION}.
+You are provided with two prompt versions below, where the second version performs better than the first one.
+[Worse prompt]
+{WORSE_PROMPT_TEXT}
+[Better prompt]
+{BETTER_PROMPT_TEXT}
+You respond only with one small hint for designing better prompts , based on the two prompt versions and using less than 20 words.
+I want you to generate only one new hint. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_NO_ROLE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts.
+
+Below is your prior long-term reflection on designing prompts for {PROBLEM_DESCRIPTION}.
+{PRIOR_LONG_TERM_REFLECTION}
+
+Below are some newly gained insights.
+{NEW_SHORT_TERM_REFLECTIONS}
+
+Write the constructive hint for designing better prompts, based on prior reflections and new insights and using less than 50 words.
+I want you to generate only one new constructive hint. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_NO_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems.
+Your response outputs prompt text and nothing else.
+
+Write a prompt for the task: {PROBLEM_DESCRIPTION}.
+
+[Worse prompt]
+{WORSE_PROMPT_TEXT}
+[Better prompt]
+{BETTER_PROMPT_TEXT}
+[Reflection]
+{SHORT_TERM_REFLECTION}
+[Improved prompt]
+Please write an improved prompt, according to the reflection.
+Bracket the final prompt with .
+"""
+
+REFLECTIVEPROMPT_MUTATION_TEMPLATE_NO_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems.
+Your response outputs prompt text and nothing else.
+
+Write a prompt for {PROBLEM_DESCRIPTION}.
+[Prior reflection]
+{LONG_TERM_REFLECTION}
+[Prompt]
+{ELITIST_PROMPT_TEXT}
+[Improved prompt]
+Please write a mutated prompt, according to the reflection.
+Output prompt only.
+Bracket the final prompt with .
+"""
+
+REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_NO_ROLE = """Paraphrase the given prompt text keeping its initial meaning.
+Prompt: {PROMPT}
+Create the new variations of this prompt and output them in JSON structure below:
+{{
+ "prompts": [
+ "New prompt 1",
+ "New prompt 2",
+ "New prompt 3",
+ ...
+ "New prompt {NUM_PROMPTS}",
+ ]
+}}
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_NO_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems.
+Write a prompt that will effectively solve the task: {PROBLEM_DESCRIPTION}.
+Output prompt only.
+Bracket the final prompt with .
+"""
diff --git a/coolprompt/utils/prompt_templates/reflective_templates_orig.py b/coolprompt/utils/prompt_templates/reflective_templates_orig.py
new file mode 100644
index 00000000..a6393b7b
--- /dev/null
+++ b/coolprompt/utils/prompt_templates/reflective_templates_orig.py
@@ -0,0 +1,76 @@
+REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts.
+
+Below are two prompts for {PROBLEM_DESCRIPTION}.
+You are provided with two prompt versions below, where the second version performs better than the first one.
+[Worse prompt]
+{WORSE_PROMPT}
+[Better prompt]
+{BETTER_PROMPT}
+You respond only with one small hint for designing better prompts , based on the two prompt versions and using less than 20 words.
+I want you to generate only one new hint. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts.
+
+Below is your prior longβterm reflection on designing prompts for {PROBLEM_DESCRIPTION}.
+{PRIOR_LONG_TERM_REFLECTION}
+
+Below are some newly gained insights.
+{NEW_SHORT_TERM_REFLECTIONS}
+
+Write the constructive hint for designing better prompts, based on prior reflections and new insights and using less than 50 words.
+I want you to generate only one new constructive hint. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_CROSSOVER_TEMPLATE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems.
+Your response outputs prompt text and nothing else.
+
+Write a prompt for the task: {PROBLEM_DESCRIPTION}.
+
+[Worse prompt]
+{WORSE_PROMPT}
+[Better prompt]
+{BETTER_PROMPT}
+[Reflection]
+{SHORT_TERM_REFLECTION}
+[Improved prompt]
+Please write an improved prompt, according to the reflection.
+Bracket the final prompt with .
+"""
+
+REFLECTIVEPROMPT_MUTATION_TEMPLATE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems.
+Your response outputs prompt text and nothing else.
+
+Write a prompt for {PROBLEM_DESCRIPTION}.
+[Prior reflection]
+{LONG_TERM_REFLECTION}
+[Prompt]
+{ELITIST_PROMPT}
+[Improved prompt]
+Please write a mutated prompt, according to the reflection.
+Output prompt only.
+Bracket the final prompt with .
+"""
+
+REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE = """Paraphrase the given prompt text keeping its initial meaning.
+Prompt: {PROMPT}
+Create the new variations of this prompt and output them in JSON structure below:
+{{
+ "prompts": [
+ "New prompt 1",
+ "New prompt 2",
+ "New prompt 3",
+ ...
+ "New prompt {NUM_PROMPTS}",
+ ]
+}}
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems.
+Write a prompt that will effectively solve the task: {PROBLEM_DESCRIPTION}.
+Output prompt only.
+Bracket the final prompt with .
+"""
diff --git a/coolprompt/utils/prompt_templates/reflective_templates_text_only.py b/coolprompt/utils/prompt_templates/reflective_templates_text_only.py
new file mode 100644
index 00000000..6b2bfa8a
--- /dev/null
+++ b/coolprompt/utils/prompt_templates/reflective_templates_text_only.py
@@ -0,0 +1,113 @@
+REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_TEXT_ONLY = """Create {NUM_PROMPTS} concise variations of the task instruction below.
+
+Task instruction: {PROMPT}
+
+Rules:
+- Each variation must be 1-2 sentences only.
+- Preserve the output format requirement exactly (e.g. numeric label, single word, etc.).
+- Change wording, emphasis, or directness β do not add extra sentences or explanations.
+- Do not include any system role or behavior description in the instruction.
+
+Output JSON:
+{{
+ "prompts": [
+ "Variation 1",
+ "Variation 2",
+ ...
+ "Variation {NUM_PROMPTS}"
+ ]
+}}
+Output JSON data only.
+"""
+
+REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_TEXT_ONLY = """You are an expert in prompt optimization. Give a brief hint for writing better task instructions.
+
+Task: {PROBLEM_DESCRIPTION}
+
+Two task instructions were tested.
+[Worse instruction] (score: {WORSE_SCORE})
+{WORSE_PROMPT_TEXT}
+[Better instruction] (score: {BETTER_SCORE})
+{BETTER_PROMPT_TEXT}
+
+Why does the better instruction lead to higher scores? Focus on wording, directness, or clarity differences.
+Give one actionable hint in less than 20 words.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_TEXT_ONLY = """You are an expert in prompt optimization. Synthesize hints for writing better task instructions.
+
+Task: {PROBLEM_DESCRIPTION}
+
+Best-performing task instructions found so far (ranked by score, best first):
+{TOP_PROMPTS_HISTORY}
+
+Study these top instructions carefully: what phrasing, verb choice, or framing distinguishes the higher-scoring ones? What do they have in common that lower-scoring instructions lack?
+
+Prior accumulated insight:
+{PRIOR_LONG_TERM_REFLECTION}
+
+New observations from recent comparisons:
+{NEW_SHORT_TERM_REFLECTIONS}
+
+Write one updated actionable hint (less than 40 words) about what makes a task instruction score higher on this specific task.
+Bracket the final hint with .
+"""
+
+REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_TEXT_ONLY = """You are an expert in prompt optimization. Combine two task instructions into a better one.
+
+Task: {PROBLEM_DESCRIPTION}
+
+[Worse instruction]
+{WORSE_PROMPT_TEXT}
+[Better instruction]
+{BETTER_PROMPT_TEXT}
+[Reflection]
+{SHORT_TERM_REFLECTION}
+
+Write a new, improved task instruction.
+Rules:
+- 1-2 sentences only. No role or behavior description.
+- Keep the output format requirement (how the answer must be returned) from the better instruction.
+- No motivational or filler language.
+Bracket the final instruction with .
+"""
+
+REFLECTIVEPROMPT_MUTATION_TEMPLATE_TEXT_ONLY = """You are an expert in prompt optimization. Improve the task instruction below.
+
+Task: {PROBLEM_DESCRIPTION}
+
+[Accumulated insight on what works for this task]
+{LONG_TERM_REFLECTION}
+
+[Current best task instruction] (score: {ELITIST_SCORE})
+{ELITIST_PROMPT_TEXT}
+
+[Examples where the current instruction most often fails]
+Each example shows the input, the wrong answer the model gave, and the correct answer.
+{BAD_EXAMPLES}
+
+Analyze these failure cases before writing:
+- What type of inputs does the model get wrong?
+- Is there a common pattern (e.g., ambiguous phrasing, specific input type, edge case)?
+- What does the correct answer reveal about what the instruction fails to convey?
+
+Write a mutated instruction that directly addresses the identified failure pattern.
+Rules:
+- 1-2 sentences only. No role or behavior description.
+- Keep the output format requirement intact (e.g. numeric label, exact phrasing for the answer format).
+- Do not add motivational phrases, emotional language, or explanations targeted at the reader.
+- Must be meaningfully different from the current instruction.
+Bracket the final instruction with .
+"""
+
+REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_TEXT_ONLY = """Write a concise task instruction for the following task.
+
+Task: {PROBLEM_DESCRIPTION}
+
+Rules:
+- 1-2 sentences only. No role or behavior description.
+- Include how the answer should be returned (output format).
+- Be direct and clear.
+Bracket the final instruction with .
+"""
diff --git a/coolprompt/utils/var_validation.py b/coolprompt/utils/var_validation.py
index f5195245..4d76f55b 100644
--- a/coolprompt/utils/var_validation.py
+++ b/coolprompt/utils/var_validation.py
@@ -8,7 +8,10 @@
from coolprompt.optimizer.hyper.meta_prompt import HyPERLightMethod
from coolprompt.optimizer.hyper.hyper import HyPERMethod
from coolprompt.optimizer.prompt_compressor import CompressorMethod
-from coolprompt.optimizer.reflective_prompt import ReflectiveMethod
+from coolprompt.optimizer.reflective_prompt import (
+ ReflectiveMethod,
+ CoevoMethod,
+)
from coolprompt.optimizer.regps import ReGPSMethod
from coolprompt.optimizer.rider import RIDERGenesisMethod
from coolprompt.utils.enums import PD_Method, Task
@@ -18,6 +21,7 @@
"hyper_light": HyPERLightMethod,
"hyper": HyPERMethod,
"reflective": ReflectiveMethod,
+ "coevo": CoevoMethod,
"distill": DistillMethod,
"regps": ReGPSMethod,
"compress": CompressorMethod,
diff --git a/notebooks/examples/benchmark_coevo.png b/notebooks/examples/benchmark_coevo.png
new file mode 100644
index 00000000..a479816d
Binary files /dev/null and b/notebooks/examples/benchmark_coevo.png differ
diff --git a/notebooks/examples/coevo_demo.ipynb b/notebooks/examples/coevo_demo.ipynb
new file mode 100644
index 00000000..b6bce120
--- /dev/null
+++ b/notebooks/examples/coevo_demo.ipynb
@@ -0,0 +1,1058 @@
+{
+ "cells": [
+ {
+ "cell_type": "markdown",
+ "id": "b2e9afbd",
+ "metadata": {},
+ "source": [
+ "# CoEvo \u2014 \u043a\u0430\u043a \u043c\u0435\u0442\u043e\u0434 \u0443\u043b\u0443\u0447\u0448\u0430\u0435\u0442 \u043f\u0440\u043e\u043c\u043f\u0442\n",
+ "\n",
+ "CoEvo \u0430\u0432\u0442\u043e\u043c\u0430\u0442\u0438\u0447\u0435\u0441\u043a\u0438 \u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0438\u0440\u0443\u0435\u0442 \u043f\u0440\u043e\u043c\u043f\u0442: \u0440\u0430\u0441\u043a\u043b\u0430\u0434\u044b\u0432\u0430\u0435\u0442 \u0435\u0433\u043e \u043d\u0430 \u0442\u0440\u0438 \u043f\u043e\u043b\u044f \u2014 **\u0440\u043e\u043b\u044c**, **\u043e\u043f\u0438\u0441\u0430\u043d\u0438\u0435\n",
+ "\u0437\u0430\u0434\u0430\u0447\u0438** \u0438 **\u043e\u0433\u0440\u0430\u043d\u0438\u0447\u0435\u043d\u0438\u044f \u043d\u0430 \u043e\u0442\u0432\u0435\u0442** \u2014 \u0438 \u044d\u0432\u043e\u043b\u044e\u0446\u0438\u043e\u043d\u0438\u0440\u0443\u0435\u0442 \u0438\u0445 \u0432\u043c\u0435\u0441\u0442\u0435, \u0438\u0441\u043f\u043e\u043b\u044c\u0437\u0443\u044f \u0440\u0435\u0444\u043b\u0435\u043a\u0441\u0438\u044e (\u043c\u043e\u0434\u0435\u043b\u044c\n",
+ "\u0441\u0430\u043c\u0430 \u0430\u043d\u0430\u043b\u0438\u0437\u0438\u0440\u0443\u0435\u0442, \u0447\u0435\u043c \u0443\u0434\u0430\u0447\u043d\u044b\u0435 \u043f\u0440\u043e\u043c\u043f\u0442\u044b \u043e\u0442\u043b\u0438\u0447\u0430\u044e\u0442\u0441\u044f \u043e\u0442 \u043d\u0435\u0443\u0434\u0430\u0447\u043d\u044b\u0445).\n",
+ "\n",
+ "\u0417\u0434\u0435\u0441\u044c \u0431\u0435\u0440\u0451\u043c \u043f\u0440\u043e\u0441\u0442\u043e\u0439 \u043f\u0440\u043e\u043c\u043f\u0442 \u0434\u043b\u044f \u0437\u0430\u0434\u0430\u0447\u0438 \u00ab\u043e\u0442\u0432\u0435\u0442\u044c \u043d\u0430 \u0432\u043e\u043f\u0440\u043e\u0441 \u043f\u043e \u043a\u043e\u043d\u0442\u0435\u043a\u0441\u0442\u0443\u00bb \u0438 \u0441\u043c\u043e\u0442\u0440\u0438\u043c, \u043d\u0430\u0441\u043a\u043e\u043b\u044c\u043a\u043e\n",
+ "\u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0430\u0446\u0438\u044f \u043f\u043e\u0434\u043d\u0438\u043c\u0430\u0435\u0442 \u043a\u0430\u0447\u0435\u0441\u0442\u0432\u043e (BERTScore).\n",
+ "\n",
+ "> \u0414\u043b\u044f \u0437\u0430\u043f\u0443\u0441\u043a\u0430 \u043d\u0443\u0436\u0435\u043d `OPENAI_API_KEY` \u0438 `pip install coolprompt datasets`."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 1,
+ "id": "063da7ad",
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-07-24T09:54:13.265675Z",
+ "iopub.status.busy": "2026-07-24T09:54:13.265509Z",
+ "iopub.status.idle": "2026-07-24T09:54:20.287984Z",
+ "shell.execute_reply": "2026-07-24T09:54:20.287556Z"
+ }
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "\u0433\u043e\u0442\u043e\u0432\u043e\n"
+ ]
+ }
+ ],
+ "source": [
+ "import os\n",
+ "from pathlib import Path\n",
+ "# \u043f\u043e\u0434\u0445\u0432\u0430\u0442\u044b\u0432\u0430\u0435\u043c \u043b\u043e\u043a\u0430\u043b\u044c\u043d\u044b\u0439 .env, \u0435\u0441\u043b\u0438 \u043e\u043d \u0435\u0441\u0442\u044c (\u0438\u043d\u0430\u0447\u0435 \u0434\u043e\u0441\u0442\u0430\u0442\u043e\u0447\u043d\u043e \u043f\u0435\u0440\u0435\u043c\u0435\u043d\u043d\u043e\u0439 \u043e\u043a\u0440\u0443\u0436\u0435\u043d\u0438\u044f)\n",
+ "for _p in [Path.cwd(), *Path.cwd().parents]:\n",
+ " _env = _p / \".env\"\n",
+ " if _env.exists():\n",
+ " for _line in _env.read_text().splitlines():\n",
+ " if \"=\" in _line and not _line.lstrip().startswith(\"#\"):\n",
+ " _k, _v = _line.split(\"=\", 1); os.environ.setdefault(_k.strip(), _v.strip())\n",
+ " break\n",
+ "assert os.environ.get(\"OPENAI_API_KEY\"), \"\u041d\u0443\u0436\u0435\u043d OPENAI_API_KEY (\u0438\u043b\u0438 \u043b\u043e\u043a\u0430\u043b\u044c\u043d\u044b\u0439 .env)\"\n",
+ "\n",
+ "import logging, warnings\n",
+ "import matplotlib.pyplot as plt\n",
+ "from datasets import load_dataset\n",
+ "from langchain_openai import ChatOpenAI\n",
+ "from coolprompt.assistant import PromptTuner\n",
+ "from coolprompt.utils.logging_config import logger\n",
+ "for _h in logger.handlers:\n",
+ " _h.setLevel(logging.ERROR)\n",
+ "logger.propagate = False\n",
+ "warnings.filterwarnings(\"ignore\")\n",
+ "print(\"\u0433\u043e\u0442\u043e\u0432\u043e\")"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "84062783",
+ "metadata": {},
+ "source": [
+ "## \u0414\u0430\u043d\u043d\u044b\u0435\n",
+ "\n",
+ "\u041f\u0430\u0440\u044b \u00ab\u0432\u043e\u043f\u0440\u043e\u0441 + \u043a\u043e\u043d\u0442\u0435\u043a\u0441\u0442 \u2192 \u043e\u0442\u0432\u0435\u0442\u00bb \u0438\u0437 \u043e\u0442\u043a\u0440\u044b\u0442\u043e\u0433\u043e QA-\u043d\u0430\u0431\u043e\u0440\u0430. \u041e\u0442\u0432\u0435\u0442\u044b \u043a\u043e\u0440\u043e\u0442\u043a\u0438\u0435, \u043f\u043e\u044d\u0442\u043e\u043c\u0443 \u043f\u0440\u043e\u043c\u043f\u0442,\n",
+ "\u043a\u043e\u0442\u043e\u0440\u044b\u0439 \u044d\u0442\u043e\u0433\u043e \u043d\u0435 \u0443\u0447\u0438\u0442\u044b\u0432\u0430\u0435\u0442, \u043b\u0435\u0433\u043a\u043e \u0442\u0435\u0440\u044f\u0435\u0442 \u0432 \u043a\u0430\u0447\u0435\u0441\u0442\u0432\u0435 \u2014 \u0435\u0441\u0442\u044c \u043a\u0443\u0434\u0430 \u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0438\u0440\u043e\u0432\u0430\u0442\u044c."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 2,
+ "id": "d8ac94a9",
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-07-24T09:54:20.289557Z",
+ "iopub.status.busy": "2026-07-24T09:54:20.289337Z",
+ "iopub.status.idle": "2026-07-24T09:54:23.020457Z",
+ "shell.execute_reply": "2026-07-24T09:54:23.019440Z"
+ }
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "\u043f\u0440\u0438\u043c\u0435\u0440\u043e\u0432: 60\n",
+ "\n",
+ "\u0432\u0445\u043e\u0434 : Context: The Normans (Norman: Nourmands; French: Normands; Latin: Normanni) were the people who in the 10th and 11th centuries gave their name to Norm ...\n",
+ "\u043e\u0442\u0432\u0435\u0442: France\n"
+ ]
+ }
+ ],
+ "source": [
+ "ds = load_dataset(\"rajpurkar/squad_v2\", split=\"validation\").filter(\n",
+ " lambda r: len(r[\"answers\"][\"text\"]) > 0)\n",
+ "rows = ds.select(list(range(0, 5000, 80))[:60])\n",
+ "dataset = [f\"Context: {r['context']}\\nQuestion: {r['question']}\" for r in rows]\n",
+ "target = [r[\"answers\"][\"text\"][0] for r in rows]\n",
+ "\n",
+ "print(\"\u043f\u0440\u0438\u043c\u0435\u0440\u043e\u0432:\", len(dataset))\n",
+ "print(\"\\n\u0432\u0445\u043e\u0434 :\", dataset[0][:150], \"...\")\n",
+ "print(\"\u043e\u0442\u0432\u0435\u0442:\", target[0])"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "11d82975",
+ "metadata": {},
+ "source": [
+ "## \u0417\u0430\u043f\u0443\u0441\u043a CoEvo\n",
+ "\n",
+ "\u0421\u0442\u0430\u0440\u0442\u0443\u0435\u043c \u0441 \u043e\u0431\u044b\u0447\u043d\u043e\u0433\u043e \u043f\u0440\u043e\u0441\u0442\u043e\u0433\u043e \u043f\u0440\u043e\u043c\u043f\u0442\u0430. `PromptTuner.run` \u0441\u0430\u043c \u0437\u0430\u043c\u0435\u0440\u0438\u0442 \u0435\u0433\u043e \u043a\u0430\u0447\u0435\u0441\u0442\u0432\u043e (baseline),\n",
+ "\u043f\u0440\u043e\u0432\u0435\u0434\u0451\u0442 \u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0430\u0446\u0438\u044e \u0438 \u0437\u0430\u043c\u0435\u0440\u0438\u0442 \u0438\u0442\u043e\u0433. \u0426\u0435\u043b\u0435\u0432\u0430\u044f \u043c\u043e\u0434\u0435\u043b\u044c \u2014 `gpt-4.1-nano`, \u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0430\u0442\u043e\u0440 \u2014 `gpt-4o-mini`.\n",
+ "\u041f\u043e\u043f\u0443\u0442\u043d\u043e \u0441\u043e\u0431\u0438\u0440\u0430\u0435\u043c \u043b\u0443\u0447\u0448\u0438\u0439 \u0440\u0435\u0437\u0443\u043b\u044c\u0442\u0430\u0442 \u043f\u043e \u044d\u043f\u043e\u0445\u0430\u043c \u0434\u043b\u044f \u0433\u0440\u0430\u0444\u0438\u043a\u0430."
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 3,
+ "id": "b2b4a75f",
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-07-24T09:54:23.021695Z",
+ "iopub.status.busy": "2026-07-24T09:54:23.021600Z",
+ "iopub.status.idle": "2026-07-24T10:05:44.980594Z",
+ "shell.execute_reply": "2026-07-24T10:05:44.979980Z"
+ }
+ },
+ "outputs": [
+ {
+ "output_type": "stream",
+ "name": "stdout",
+ "text": [
+ "\u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0430\u0446\u0438\u044f \u0437\u0430\u0432\u0435\u0440\u0448\u0435\u043d\u0430\n"
+ ]
+ }
+ ],
+ "source": [
+ "START_PROMPT = \"Answer the question based on the context.\"\n",
+ "\n",
+ "target_model = ChatOpenAI(model=\"gpt-4.1-nano\", temperature=0, max_tokens=96, timeout=60, max_retries=3)\n",
+ "optimizer_model = ChatOpenAI(model=\"gpt-4o-mini\", temperature=0.7, max_tokens=1500, timeout=90, max_retries=3)\n",
+ "\n",
+ "tuner = PromptTuner(target_model=target_model, system_model=optimizer_model)\n",
+ "tuner.run(\n",
+ " start_prompt=START_PROMPT,\n",
+ " task=\"generation\",\n",
+ " metric=\"bertscore\",\n",
+ " dataset=dataset,\n",
+ " target=target,\n",
+ " method=\"coevo\",\n",
+ " system_model_as_optimizer=True,\n",
+ " validation_size=0.34,\n",
+ " population_size=4,\n",
+ " num_epochs=5,\n",
+ " verbose=0,\n",
+ ")\n",
+ "print(\"\u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0430\u0446\u0438\u044f \u0437\u0430\u0432\u0435\u0440\u0448\u0435\u043d\u0430\")"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "7099f268",
+ "metadata": {},
+ "source": [
+ "## \u0420\u0435\u0437\u0443\u043b\u044c\u0442\u0430\u0442: \u0431\u044b\u043b\u043e / \u0441\u0442\u0430\u043b\u043e"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 4,
+ "id": "23f75979",
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-07-24T10:05:44.986456Z",
+ "iopub.status.busy": "2026-07-24T10:05:44.986350Z",
+ "iopub.status.idle": "2026-07-24T10:05:44.990288Z",
+ "shell.execute_reply": "2026-07-24T10:05:44.989876Z"
+ }
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "\u0441\u0442\u0430\u0440\u0442\u043e\u0432\u044b\u0439 \u043f\u0440\u043e\u043c\u043f\u0442 : BERTScore = 0.8233\n",
+ "\u043f\u043e\u0441\u043b\u0435 CoEvo : BERTScore = 0.8963\n",
+ "\u043f\u0440\u0438\u0440\u043e\u0441\u0442 : +0.0730\n"
+ ]
+ }
+ ],
+ "source": [
+ "init_score = tuner.init_metric\n",
+ "final_score = tuner.final_metric\n",
+ "print(f\"\u0441\u0442\u0430\u0440\u0442\u043e\u0432\u044b\u0439 \u043f\u0440\u043e\u043c\u043f\u0442 : BERTScore = {init_score:.4f}\")\n",
+ "print(f\"\u043f\u043e\u0441\u043b\u0435 CoEvo : BERTScore = {final_score:.4f}\")\n",
+ "print(f\"\u043f\u0440\u0438\u0440\u043e\u0441\u0442 : {final_score-init_score:+.4f}\")"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 5,
+ "id": "d753197f",
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-07-24T10:05:44.991427Z",
+ "iopub.status.busy": "2026-07-24T10:05:44.991333Z",
+ "iopub.status.idle": "2026-07-24T10:05:44.994396Z",
+ "shell.execute_reply": "2026-07-24T10:05:44.994093Z"
+ }
+ },
+ "outputs": [
+ {
+ "name": "stdout",
+ "output_type": "stream",
+ "text": [
+ "\u0411\u044b\u043b\u043e \u2014 \u043e\u0434\u0438\u043d \u043f\u0440\u043e\u043c\u043f\u0442:\n",
+ " Answer the question based on the context. \n",
+ "\n",
+ "\u0421\u0442\u0430\u043b\u043e \u2014 \u0442\u0440\u0438 \u043f\u043e\u043b\u044f, \u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u043f\u043e\u0434\u043e\u0431\u0440\u0430\u043b CoEvo:\n",
+ " \u0440\u043e\u043b\u044c : You are a focused extractor; ensure the answer is directly from the context.\n",
+ " \u0437\u0430\u0434\u0430\u0447\u0430 : Extract the precise answer from the context provided. Respond in JSON format.\n",
+ " \u043e\u0433\u0440\u0430\u043d\u0438\u0447\u0435\u043d\u0438\u044f: Response must be in JSON format, limited to 100 characters, with only the answer as the value.\n"
+ ]
+ }
+ ],
+ "source": [
+ "print(\"\u0411\u044b\u043b\u043e \u2014 \u043e\u0434\u0438\u043d \u043f\u0440\u043e\u043c\u043f\u0442:\")\n",
+ "print(\" \", tuner.init_prompt, \"\\n\")\n",
+ "print(\"\u0421\u0442\u0430\u043b\u043e \u2014 \u0442\u0440\u0438 \u043f\u043e\u043b\u044f, \u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u043f\u043e\u0434\u043e\u0431\u0440\u0430\u043b CoEvo:\")\n",
+ "print(\" \u0440\u043e\u043b\u044c :\", tuner.final_role)\n",
+ "print(\" \u0437\u0430\u0434\u0430\u0447\u0430 :\", tuner.final_prompt)\n",
+ "print(\" \u043e\u0433\u0440\u0430\u043d\u0438\u0447\u0435\u043d\u0438\u044f:\", tuner.final_constraints)"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "9ead0ba1",
+ "metadata": {},
+ "source": [
+ "## \u0412\u0438\u0437\u0443\u0430\u043b\u0438\u0437\u0430\u0446\u0438\u044f"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": 6,
+ "id": "f3273c64",
+ "metadata": {
+ "execution": {
+ "iopub.execute_input": "2026-07-24T10:05:44.995486Z",
+ "iopub.status.busy": "2026-07-24T10:05:44.995430Z",
+ "iopub.status.idle": "2026-07-24T10:05:45.228724Z",
+ "shell.execute_reply": "2026-07-24T10:05:45.228222Z"
+ }
+ },
+ "outputs": [
+ {
+ "output_type": "display_data",
+ "data": {
+ "image/png": "iVBORw0KGgoAAAANSUhEUgAAA38AAAJYCAYAAADSaV+1AAAAOnRFWHRTb2Z0d2FyZQBNYXRwbG90bGliIHZlcnNpb24zLjExLjEsIGh0dHBzOi8vbWF0cGxvdGxpYi5vcmcvctoD+AAAAAlwSFlzAAAViAAAFYgBxNdAoAAAXzhJREFUeJzt3QeUU1X39/FN770MvfciXXoTFAtFBUVQAQUExUZReFQUFfRRARVUFEQQpFhABJSqgNSHooJIUar03oZe8q593pX8b27KJDOZEu73s1ZWJjc3NyfJTDK/nHP2SeVyuVwCAAAAALippU7uBgAAAAAAEh/hDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAB0ib3A0AcPPZt2+fHD9+XC5cuCA5cuSQ0qVLS6ZMmRLlvtasWSOXLl0Kad9s2bJJrVq1xAmuXLkix44dM6cjR45Izpw5pW7dusndrBTp8uXLsnr1avNzhQoVpECBAsndJEfbtGmTnDx50vysr4W+JoHcuHFDdu7cKSdOnJA0adJI7ty5JW/evOZ9J1wXL1407116rAwZMki+fPmkaNGikhz+/PNP0w6rYsWKSalSpcJ6P0yVKpWkTZvWPB59DyhYsKBkyZJFUpKjR4/K/v37zeeFtk0fY7DX7+rVq7Jy5UrP5erVq5vHBiBELgCIgNWrV7s6derkypcvn0vfWqyn1KlTu2699VbXe++95zp16lREn+/ixYv73F+gU61atVw3o7Nnz7pGjRrl6tixo6tGjRquvHnz+jz2qlWrJnczU6zvv//e8zz9/vvvyd0cRzt48KArc+bMntfj66+/9rvfv//+63rsscdc2bJl8/u3XqBAAdc999zj+uijj4Le37Vr11xTpkxxNW7c2JU+fXqf48TExLi6devm2rx5syupXL161ZU/f36ftjRo0CAi74dFixY1z93atWtdyeHixYuuMWPGuB588EG/nxepUqUy79VTp04NeIzatWt79n/yySeTtP1AtCP8AUiQ2NhY16OPPhpyABs9enREn3Gnh799+/a5ypYtG9Lj37RpU3I3N0Xq2rWreX5KlCiR3E1xvCeeeMLz+1q6dGkTzuz++OMPV65cuRL8N79//35X3bp1QzqOfoH1+uuvu27cuJHor9GcOXMCtuOff/6J2PuhnvQLu3PnzrmS0u7du0NuX/fu3f0e49tvv/XskzZtWtfff/+dpI8BiGYM+wSQoOFyd955p6xYscLnupIlS5ohRqdOnZLt27eb4VlJQYcMBRqqVb58ebmZ6Bd4HTt2lH/++cezTYe9FS9eXPLnzy8xMTFSqFAhKVKkiJQoUcKcw9v169flxx9/ND+3a9eOpycZ7dq1S7744gvP5eeff94M5bTr1q2beV+x/93r77sOldS/B/3bCEaHlTZp0sTcp5UON9RhprGxsbJlyxbPcfT967XXXjNDQ99++21JTJMmTQp43VdffSVDhgwJ+/1Q2/3vv//K4cOHva6fNm2a/P333/Lrr79K5syZJTno8Fpt54EDB8zwT6vx48dLy5Yt5aGHHvLafv/995vPmN27d8u1a9fMczJlypQkbjkQnQh/AOLtxRdf9Al+LVq0kNGjR0vFihU923T+37hx4+Sdd94Jejz9R0s/zHWeWurUqc18n3Dn3Dz55JMyYMCAOPfTf3gOHjzouVy2bFkpXLiwz36HDh0y4dVNA62/EBmftuv9azvc9J9X6/MWl++++05WrVplfq5du7Z8+umnJuDpP3n6D1GePHnMfEt//0C7g8/y5cs9l7Xd+g9xMHpc62uux27cuLHffXU/3T8uOiepUaNGifp7EYi2UX8/1b333hvv4+jviZ708WrQ1vBtpXMK9cuScPiby6T/IOvvjc7p1Dms+g+wnocipb8eY8eO9bRPf68efPBBn322bt0qf/zxh+eyPj8LFy6UOnXqeLbpHFcNSSNHjgx4X7179/YKfjo3bujQofLCCy9IunTpzDadS/jwww/L//73P89+//3vf80XXk2bNvVs06C4fv16z2Wd32yfX7tjxw6vYKN/l/6eszNnzsicOXM8l/X5tX5xNnny5LDCn/39cOPGjTJo0CCZP3++Z9uGDRukT58+MmHChJCOGan3Tg11L730knku9XEqfd26du3q9ZgnTpzoE/50f/3iS18P93vhBx98YIIkgDgkd9cjgOi0Z88eV7p06byG6Nx2221mvkqw4T5LlizxO8/n6aefduXJk8dn2E+RIkVcr732WsChSfZhTjqvMBRfffWVz/Anf3S+j3W/8ePHR6ztOu/Fuv/DDz/sCkfbtm09t33qqadc5cqV82lD9uzZzWPYuXOnz+21XdZ9M2TIEOd96pxN622yZMkScN8cOXKENLRL9/MnIc9tqPr27WuOlzt3br9DDON6LgYPHuwqVqyYT/tKlizpevPNN12XLl0y+xYuXDis4Xh6cv+tfPPNN+Z3w99cTp0fVbNmTfP7HJeU/HpcuXLFzK9zH69ly5YhDYl86KGHAh5T2zJ27Fif7Vu2bDHPm/U4AwYMCPga6/xB6772tuk8Uev1OlzV7rnnngvpfUrba92vTZs2rjJlynhtW7FiRcDHHMr7oQ5d7dChg8+w1r/++suVFO+dR44ccU2fPj3g8Tt37uzzt+SP/Xl/9913Q2o/4HSEPwDxMmLECJ9/QkP958Hqf//7n99J//ZThQoVzPy2uP7Z6d27t/mn2d/p8OHDnttduHDB659hDTHnz5/3Orb+427fx/rPbULbntDwp/9whxoktDDGggULoib8JfS5DVWpUqXMcbp06RLW7bQAiL/QZz+525aQ8BfqbQMFmGh4PRYtWuR1HH1/8Ud/h637VatWLegXTv4MGzbM6xj6Jdbx48dD3j9NmjSmyFKgEKJhLb7hr1GjRl77TZo0yfWf//zHa1uvXr0CtjXUL8M0yNsL3OgXGaGIxHtnMG+99VbIxaoKFix4U8/pBhID6/wBiBfrcEFVqVIlcwqHzttp27atGT7mpsO99Dj2kubbtm2T++67L865gzr0sXnz5n5PixYt8hqaZR1KdP78eZk9e7bXsX766SczDMvtgQcekKxZsyZa28NlvW97SfiqVaua4Xtu586dkw4dOsjevXslqdgfb4MGDcwQr/r16we9XVI9t1pO3z30L5whn/pc3nXXXWZ4rVX27NnNUiL+htrqY9bH7j7Z/1Z0iK71ej3Zh3zq76w+fj2Wvr7p06f3un7EiBHmMUXj66FzzqxuvfVWv/vp8+selukexqjPxVtvvSVLliwJ+DdhXw7BqkaNGub5D0SHstuHS69bty7g/jqEND50KK11CQN9ffV51/cdq2+++SbsIcR2OgRTX/9g7+mBJPS9My6bN2/2umxvp5V1uK8OBz579mxI9wE4WqJESgA3PXuVvPbt24d9jCFDhngdQ4ferVu3zqsEvw5Hsu4zY8aMeFe3mzx5stdt16xZ43W9DqO0euCBB7yu//XXXyPa9h9++MHVtGlTz2no0KFhPX/WYXLuk3WYmw5vs/caWcuiJ3bPn15n3ffkyZNm+6FDh4L2NEXiuQ3FG2+8YW6bKVMmn56LYOw9QdrrPXz4cK9ho+7lN44dOxZntUI9tWvXLuD9ac/P0qVLfSpNnj592gxBtB5Hh15G4+uhQ8atz2ew12PgwIFB/851mKD+nm/YsMHv7XXZGev++ncejI4YsN/HtGnTAvb8ae9nfHr+tJqodZ/WrVt7rtOhpNbrvvvuuwQPg+/Zs6fXvuXLl3eFKiHvncHo75VW77T2yur7WFx/w+7T/PnzQ34MgFPR8wcgXrR6nJUuIhwu+7fFzz33nClc4qa9MVrVzcpaDMEf7Ymw96C4T1pQxUqLMlSuXNlzWYsguKsIahGHuXPneq4rU6aMV2GTSLRdv9VfunSp5/Tyyy9LuN/eW2n7evbs6bmsxWP+85//BG2DlfbWuNuiPRBa7VALi8SXLsZsZe2xSerfC39mzZplzm+//fawKh1+//33Xpe7d+8u/fv39+rt0yIszzzzjFlwPKG0V0t/f7UYihYO0QIky5Ytk99//92r50PpaxaNr4e1R1orbgZ7PbTaphZnCbSP9qCNGTPG9BJqW+2VP+3vXfYeVDt/723a2xWIu3hJuLTYiZW1x8/e+xesImiotAcv2PMSTELeOwPZtGmT3HPPPV5FiT788MOgRbDs74FJObIBiFaEPwDxYh8mdfTo0bCPof/IWtWrV89nH/s2+238VbezBirrSf/Jt3vsscc8P2vQmTlzpicYWP8Zsu6XWG0Pl30YVSht0IqDly5dChgO3ENktdqj/nOnQw8feeQRU2UyHBok7cEx1C8IkuK51efht99+i1eVT2uFVtWmTRtJTFrR9e677zaBUisr6mNv1qyZeZ3syw5Yh9pF0+thHa6py5UEo8Mq9YsSfQ21inDnzp1NFUl/oWvUqFFmH6tcuXJ5XT59+nScy0LY2Yfk2tsXLq0Ga12yxT3kM1D4mzdvnqdKbXxpVVSrYENf/Ynve6c/Gh71Pcf6OTJs2DDzfh6M/XcllGG/gNMR/gDEi/3bWJ0DE+48FHsI8VeyXudRWV24cEEi6dFHH/WaG6frXlnPlf5T2aVLlxTXdnvPRyhtCPcbft1X18/SJSCC9XbY2ddh07aF2tOUFM+tu9dPX9vWrVuHdVt7++IKKwmxYMEC0+un/+y7/750HceGDRv6nTuo89Gi8fWw3kfGjBlDuo2GuB49epjfT51rqI9Re8/svUH2HjUN0FbB5kkq7WG1C9Yb5W++o84TDcbek6evq85hc39xpXPZrL9n+kXN119/LfGlvaHuZWKs95kU753+5mnr36D7OdLbaGjXZSDC7b2M9HsscDMi/AGIF13ryt7joAvyxsX64Wxfk8nfkB37Nvv6aQmlx9OhRm5aNOKvv/7yKg5zxx13+CyQnhLabh9SGEobtEchUK+F9li4h8jqsD5rONDCKO5v9kOxb98+r8v2f8iDSYrn9ocffjDnGqLCXRvMvr/+viSWV1991WsYnP6jvGfPHrNmn4aCvn373hSvhzXYBOuJ0y8jAq1VqAFU1+WzrydqXzjcXsBFC/fYg5CVrq1npe8F1vBn73H0F0CsvXp22mumRVysNPjZC1bZeyATMvRTg6P9d8L+np5Y753WAKrrKmrvnvtLC/1Ca8aMGWbIdCjsz0kkhlkDNzvCH4B40WFo9m+/ddH3X375xe/++q3u888/77WQsL3KoH2RYf2nSL/Vt4qrMmF8WIcl6bf2OszROj/K37ClSLRdF0q2DkvVBazDYa+IqOHMPuzviy++8JmrE2hYmgZDd1u0J9fefuuCzaEs2G1VpUqVkG+b2L8XGi50zlx8F3a3L4CuC037qzKoISAhcybd86CsWrVq5fXPs/15iMbXwx4W/Q2ztP4O6nBkDS+BQqC9l1PnEFrpa64L01s9/fTTfnvn9H7cXxS4PfXUU15/Q/ZeT/27th5LexY1rAei8+OCPeZA1q5d6zMEORRa1dM+nFIrBD/44INJ8t7pDvFafXj48OFeX0ho1ddw/ibtz1ukv2ADbkrJXXEGQPTSCoT2hd616l/Hjh3N+lQLFy50ff31165nn33Ws0D16NGjPbfX6+1V9O69916zqLUuJFy/fn2v63RdKl1cPr7r/OnJ30Leuk6Yv8qZ7sqG7oW6rSLR9oSu86dVGjNnzux1jCpVqrgmTJjgmjlzpuvRRx8NWvHUXu1TX0v38zR37lyzwLT1el1/K65qn7/88ovP7fSkvw/WdgerLhmJ5zbURap37tzpCpdWLrS3T9dc1MqK+rxNmTLFVKTU3/lAa96FWu1Tf/+s+zVv3txU19S/K3ulTz21aNEi6l4PpessWo+xa9cuv/tZK2vq4utPPPGE+X3XKo/6vOjznjFjxoAVbt308furEjpy5EhzLK2m2bVrV58qplrJ016J9Pr162YdTet+zZo1MxVPP/roI7/vLdYqnPraW6/T9zRrFWDryb7W4ssvvxzS++G8efNcn332mbkv+2PSdQvjWyUzPu+dZ86c8am4mj17dvOaBHrftle6ddPHZz1OfNaaBZyG8AcgQbTkuf2frWAna/hTjz32WMi31X/M7MJZ6kFPgRYa7t+/v9/9n3766YCPPaFtT2j4U59++mnIbdDS8fqPaqDwF9dp/fr1cYa/Vq1a+dxOQ4r1n7e4wkYknttg3GXogy0eHZd+/fqF1LaEhj8NIMGOr8tzBAt/0fB6qHHjxnkdQ58ff+zLKsR10mU8/vnnH7/HevXVV8M6VtGiRV07duzwe6wePXoEva39PdId/nRxefsXaLqQfSDvv/++1776/md9LcN9P9Tnx74ETrjCfe/8888/w2qjnjRk+qMLu7v30S9bAoVEAP+H8AcgwTZu3Oi3F8J+ypMnj88/NvqhPmjQIJ9/gKwn/Vb9k08+8XvfkQp/+o2xv/1/++23gI87oW2PRPhTEydOdOXMmTNgG/Sbfu0huXDhgtftQg1/entdg0yFG/46depk1qOzCiVsJPS5DUR7Ity9NK+88oorvvSfzBEjRvisnWfvzTh69GiCwp/evnLlyn6Pr+FV1xIMJ/yltNfDTXsKdX0/97H69u3rd7+9e/e6qlWrFtLvbbFixczohGC0t7BcuXJBj6Pt0uct0JqNSq/TdfL83V6DoQYhf+FPewat27U309/oBOtrpT111tssW7Ys7PdDfUx33XWXee9OqHDfOyMV/nQtTevvY+fOnRP8WAAn+L8yTQAQT7fccouZ5K9z1hYvXmxK6GsZcq3gp/NttLqervN02223+VTy02pxWq5e59x89913sn79elOuW4so6LwcnTukZc4DFSnREvMlSpQIua3Wtdjsle66detm1ghzK1y4sNSoUSPgsRLa9kKFCpniKqFUEAyma9eu0r59e1MoQdfn0zlHOg9LS7dr+3UNNl1ry99zYb1/+2PTdletWtU8BnclQN1uvY292p4WitGiC1p6X+/XuhaYdW6h9Rj2JSsi8dwGonNS3fOx4jPfz03nfPXr18+s8adzLbVUvz7vOu9Jq3Hq76W2L9BadFpExfocBJqDp/vp/MuJEyeatuucTt2mf0tabVHv13qcatWqRdXr4abPmRZi0fcPpb/LI0aM8JmfqnPTtBiKFivRuWs6J1IL4Og8Tp0DqJVIda1Pfb+566674lzOQn8HdEkFnQOqc111Dp3OGdT7cC+FoM+HzmcOVkxEr9PnRJeV0HlrOqdN29GpUyfTlo8++sjrOS5atKinkJJ1uy4bEug9Sulz3bt3b9m8ebNnmz4HWo3X3/uhPn/6GunzoNVRdV5dhQoVzHNdsmRJiYRw3zuzZMkS8H0nEH/zlHX9Sev8Qv1bBBC3VJoAQ9gPAICo16tXL1P8RP/51iqPSDm04mXHjh09l/WLjAYNGiRLWzQEamEf97pxMTExpj2lS5dOlvbAl4b2OXPmmJ/1ddGKqvFZYxFwGqp9AgAcQ3v9tNdBe7CQsmjPZLly5TyXP/vss2Rri7bjp59+8vSCai+gLltw+PDhZGsT/o/2/OrC8G4DBw4k+AEhoucPAACkCN9++61nyQFdZ3LHjh1mqGdy0eHsw4YN8xpW+8EHHxA0kpmub6mvg9Ihzbq+YLDhsgD+D+EPAACkGDqX0b0Auc5nDbRWHJxJ5/ndd999Ehsbay7rfExddxZAaAh/AAAAAOAAzPkDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMJfhPXs2dOcAAAAACAlSZvcDbjZbN68ObmbAAAAAAA+6PkDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwgLTJ3QAAAAAgLrt27ZIFCxbIvn375Pr161K4cGFp0aKFVK5cOSJP3sWLF+WXX36RP//8U06cOCE3btyQnDlzSsWKFeX222+XHDlyBL39lStX5Oeff5bffvtNTp48KVmyZJFy5crJnXfeKXnz5g25Hf/++68sXrxY9u7dKxcuXJCCBQtKqVKlTBv0mEBCEP4AAACQYmkQe/rpp2X69Ol+r7/jjjtk7NixUrx48Xjfx4cffiivvvqqnD171u/1GTJkkH79+smbb74padKk8bl+ypQpMmDAADl8+HDYt7WGW93vhx9+8Ht95syZpW/fvjJ06NCwHhtglcrlcrm8tiBB6tevb85Xr17NMwkAAJAAZ86ckSZNmsimTZuC7le0aFFZsWKFFCtWLOz7eP/9903oCsVzzz0nH3zwgde2kSNHSv/+/eO87YMPPihff/213+vWr18vd911lxw/fjzoMbSnU3sFgfgi/EUY4Q8AACAyevXqZXr13HT4pG7LmDGjjB8/Xvbs2eO5rlWrVjJ//vywjq/DRwsUKOAVunQ46eOPPy7p06eXr776SrZv3+65Lm3atHLkyBHJnTu3ufzPP/+YYadXr171CmgtW7aUAwcOyOeffy6XLl3yXKeXu3fv7tUGHSJ6yy23mP3dsmfPbsJihQoVTADesWOHGZJapUoVwh8ShPAXYYQ/AACAhDt06JDpybt27ZpXD1mtWrXMz/v37zdz6nSuntu6deukdu3aId+Hzh+09xZu27ZNypcvb34+duyYGU5qvY/ly5dLo0aNzM+DBw/2GobZtGlTWbp0qddw0EceecRzuXTp0iYwpkqVyrNNew2199BNg6DObdRQag+q2gNao0aNkB8fYEe1TwAAAKQ4M2bM8Ap+GnrcwU8VKVLE9PZZBZoXGIjOo7PKlCmTJ/ipfPnymZ7AQLexD0e1t0eHclrt3LnTFISxFonRHkwr7W20Bz+l8wUJfkgowh8AAABSnLVr13pdrl69us8+9m3a8xeOPHnySKVKlTyXtYdv1apVnsvaS6dVN61hUKt/ul2+fNnreNbhn+5wZ2dt44YNG8ywTuvj0d7BadOmyaBBg+SFF16QUaNGydatW8N6XEAgVPsEAABAiqPVL6389YbFxMR4Xd69e3fY9zN69Ghp3bq1Z2inztdr06aNmfM3d+5cT6DT+X4axLR30E2DmtXMmTNl4MCBpsKnmjp1qt+hpoF6DnV+oC7roPMK7Tp27CifffZZnEtOAMHQ8wcAAIAU59y5c16XtciLnTWIKWsvWqhuu+02+d///mcCoNIQ+M0335jhl6dPnzbbGjduLMuWLZOHHnrI67adOnXyurxx40ZTAOaxxx4zQ0D9VQG1LiehxV7s8w39BT+llUK1jTr3D4gvwh8AAABSHGtRFOVvdTL7tmDr6AWiC6mPGzfOhLtAdKim9rrZw5oWftHqo/Z5fRMnTpSFCxf6PZZ1zqB9mKhKly6dPPHEE/LOO+9I27Ztva7T5SwmT54c8mMD7Bj2CQAAgBQnZ86cXpfPnz/vs499W7hDIjU8ag+dhiq3W2+91fSwaZD8+eefzRILOhxz0qRJpofwjz/+8OqF/OSTT6RkyZLy9ttv+/Q8VqtWTXLlyuVVAVSXq7Au6WD33nvvmfUErcM9tSfSbdasWdKtW7ewHifgRs8fAAAAUpyyZct6Xbaugxdomy79EA6d02cNflpNVAu+6BIOL730kgl/d955p+d6XfNPe/WsUqdObeb5HTx40KzBp2v5ffHFF6ZgjQZFa8VS93246fw+uzvuuCNoBVHrnEEgXPT8AQAAIEWunaxBKlglT/u2evXqhXUfOkfPStcItA8drVu3rtfi8fbbWIdz6gLvVlrQxVo9VHv6rG3UXkYd3modvhpXxdBs2bKF+OgAX/T8AQAAIMVp3769V0EXXXZh9uzZXsskLFmyxOs2Dz/8sE+RlKefftpz0p48K3dVTjcd4hkbG+sVvObNmxf0Nj/99JNpm532Ej7wwANy48YNz7aePXt6zfnTCqbNmjXzup21QqgWd9HHYFWnTh2f+wJCRc8fAAAAUhydv6fVMocOHerZpmFKi6DoMgwaBK2VL7t06eIz7FPDoRZqcStRooRX71zz5s299tcQV6ZMGbPcg/YAahEY6zp//m6jYW3KlClyyy23mPvXOX5a9EVva22fzgt89dVXfR7nkCFDzJxAd++fFnpZvXq1WUZC5xhu2bLFs68+bnuBGSAchD8AAACkSDr3TufOuStnak/cd99957NfjRo15IMPPgj7+DrMU5dlmDBhgmebLrWgYc4fnf9nr8BpHeJpX7fPrVixYvLjjz/6LfDSpEkTE3Bffvllz7Zff/3VnKx0eKgWl9FwCsQXwz4BAACQIrl7+DQEZs2a1ed6HYLZp08f08OnPW7xofMKtcJmwYIFA+6TO3duE8600qZ9CQqtDKpr+/mjQzx1uKkOUa1YsWLA42txGR3eqT2T/lSvXl0WLVok3bt3D/lxAf6kcvlbNAUJmpystLseAAAAkXH58mVZuXKl/Pvvv2YeXeHChaVhw4Z+Q6GbDqfcvHmz17p8GqT80X+JdV9daP3UqVPmsvbU6VBOXbIhbdrgA+b+/vtv2bFjhxw6dMiE0qJFi5riLvY5gsHoff7++++mDbrIvQZabW+4VUyBQAh/EUb4AwAAAJASMewTAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4QNrkbgAAAEBKV3jAkeRuAoBkdGB4zE3x/NPzBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA6/wBSeT8+fNy6NAhSZUqlRQsWFAyZ84c8fs4cuSInDp1Slwul+TJk0fy588f8m0vXrxo2nfhwgXJmjWrFC5cWNKlSxfy7a9duyaHDx+WM2fOSJYsWcK+PQAAABIXPX9AIps9e7Y0atRIsmfPLmXLlpUyZcqYn5s2bSrz5s1L8PF37Nghjz/+uOTNm1cKFCggFStWlEqVKklMTIzky5dPevToIbt27fK53aVLl2Tq1KnSs2dP0yYNbKVLl5aqVatKyZIlTRtbtWolixYtCnjfq1atkr59+0q5cuUkffr0UrRoUalSpYq5faZMmaROnToyatQouXLlSoIfJwAAABImlUu7CBAx9evXN+erV6/mWXU4/dN69tln5aOPPgq634ABA+S9996L1338+uuvcvfdd5texWCyZcsmCxculHr16nm2bdu2zQTFULz88ssydOhQn+0aaleuXBnn7XW/xYsXS4YMGUK6PwBIaQoPOJLcTQCQjA4Mj7kpnn96/oBEoqHPHvyKFClihkNaDR8+XD7//PN43Uf37t19gp/eh56szp07Z3r4gtFhqNqDp0HRbtiwYbJgwYKgt8+VK5dUqFBBcufO7XPdihUrZMyYMXE8GgAAACQmwh+QCDRsaW+Z1ejRo2Xfvn2yf/9+E/isXnzxRTPnLhx///23GfJpNWXKFHMfeho3bpzXdZs3bzbb/fVW69BUnSu4fft2OX36tIwYMcJnP/vxVLNmzeTLL780tzl58qRs3bpVTpw4YY5nn9P4888/h/X4AAAAEFmEPyARfPPNNyYAummP2NNPP+253L9/fylRooTnsgav77//Pqz7iI2N9bqsRWQ6d+7suaxz/ey9eNbbaDgbP368mbfXpk0bM2dPpU6dWvr162e2WWmws9OhoF26dJEcOXJ4bdfb3nfffV7bbty4EdbjAwAAQGQR/oBEsGzZMq/LTZo08dmncePGQW8Tl1KlSknatP9XsFfD5uXLl70uW3sTdb6dNXAWK1bMFIoJpHr16l6XtYBLOOy9jLVq1Qrr9gAAAIgslnoAEoEOsbQqXry4zz72bfbbxCVnzpwmvI0dO9bTq/fQQw+ZIjPay6ZFZHT5Bbc+ffqEFeB0WKlV3bp1gwY9He6pVT0PHDggX331lSlG46bVP59//vmwHh8AAAAii/AHJAKd/2alyybY2Ydk2m8TCi0oo71/Oh/v6tWrMmvWLHOy0h4/DX7vvvtuyMf9/fffZebMmZ7Lul6fddiq3euvv26GkNplzJhRevfuLS+99JLfQjAAAABIOgz7BBKBrqFnlSZNGp99rEM2/d0mFBrKHn74YVN4JZDbb7/d9Aj6a4M/ugRE69atTZh00wI1oS4LYX9MS5YskbVr14Z9WwAAAEQW4Q9IBPZePetcvEDb/PUOxkV73HQNPfdC7FqsRef16ZDSVKlSmW1z58416/vZK4z6o0GtQYMGcvDgQc+2wYMHm6Gkwej8wWrVqpnCNrrkg9XGjRulbdu28u2334b9+AAAABA5hD8gERQqVMjr8vHjx332sW/Tap3hWL16tQwZMsQsJq90/cAtW7bI7t27Zc+ePWboZp48ecx1Ogdw4MCBsmnTpoDH0yUbWrVqZSqPKg2POlT0jTfeiLMtr776qvzxxx+mIqgOX12/fr0Jg256/7qYPQAAAJIP4Q9IBPbKlv6WSbBvq1mzZlj3MW/ePK/LjzzyiJQvX95zWcNXhw4dvAJYoIXaNbx169bNM9RT5+pNnTpVXnjhBYnv47f3NP77779y7NixeB0PAAAACUfBFyAR6Dp377//vufy4sWL5cyZM5718LQypn1pBx0aaaW9d7p4unVopbVoiruHzu3w4cM+7Th06JDXZfttdOipVgzVoOeWL18+UzRGh38Go0tJ2Ie3Wu3atctn2/Xr14MeEwAAAImH8AckAi3Acsstt3iGWZ49e1buvfdeeeWVV0wPnM7Vu3Dhgmf/W2+91czLs9JhkjNmzPBcnjBhgumdc6tUqZLX/pMnTzZDP3Xopg4FnT17tjlZWW+jbbrnnntkxYoVXstHfPrpp2YBeB3GaaXDQK1DOT/88EP5/PPPpV27dqbXUu9b5y1q757OQRwzZozX7XUeYkxMTBjPIgAAACKJ8AckAg1KuvSBLu7uXmh96dKl5mSXJUsWs1RDuLSCp875O3r0qLmsofKtt94yJ3+KFi1qAqh1HT9r8FPa09i+fXu/t9dqodZ1A9XevXtl1KhRIbV32LBhniI0AAAASHrM+QMSSe3ateWnn37yKf5iD2QLFy40vYTh0qqaOu+vVKlSce6rcwHnz58vWbNmlUhxD2GNiw4j1V5JXZICAAAAyYeePyCRh39u375dpk2bZoZCHjhwwBP6dP29zp07S6ZMmfzetmTJkl7DLP0tkq7DLbVwzJw5c+SXX34xvXnae6c9bBrOdOmFli1bmuGd9nUFdWin9fhxsd/+mWeekfvvv98E3HXr1pn5hUeOHDE9hBpMy5UrZ5ahuPvuu819AQAAIHmlcrnrxCMi6tev7ynDDwAAbg6FBxxJ7iYASEYHht8cdQsY9gkAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAABwgbXI3AIlv3JfTeJoBB+vZtVNyNwEAAKQA9PwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcICoC3+7du2Sjh07Sq5cuSRdunRSqVIl+eijj8TlcoV1nEWLFkmLFi0kJiZGMmbMKBUqVJABAwbIiRMnEq3tAAAAAJBcoqra586dO6VevXpy/Phxz7atW7fKM888Y657//33QzrOBx98IH379vXatn37dnP67rvvZMOGDZInT56Itx8AAAAAkktU9fz169fPBL/WrVvL3r175cqVKyasZcmSRT788EP57bff4jzG5cuXZfDgweZnPT969KjZtnr1aqlcubI57scff5wEjwYAAAAAkk7UhL9jx47J3LlzJW/evDJ9+nQpVqyYGfbZvn17ef31182wz/Hjx8d5nH379klsbKzUqFFD3njjDcmXL5+kT5/e9CiOGDHC7LNly5YkeEQAAAAAkHSiJvytXLlSbty4IW3btjU9fVYPP/ywOV++fHmcx9E5fqlTpzbHsnNvK1SoUMTaDQAAAAApQdSEvx07dpjzKlWq+FxXoEAB0yPo3ieYbNmySbdu3WTjxo0ycOBA0xN47tw5Wbp0qTz//POm+EvPnj0T5TEAAAAAQHKJmoIvZ8+eNeda5dOf3Llzm/mAV69eNcNBg/n000+lYMGCMnr0aHn33Xc923Xo58SJE6VixYpxtqd+/fp+t2/evNlvQAUAAACA5BQ1PX9xCWeph3/++UeWLVvmCZRu27Ztk/nz5/sdEgoAAAAA0Sxqev5y5MhhzgOtw3fy5EnJlClTnL1+ul/Tpk3l4sWLppevTZs2Zijo33//bap/ahGY69evy9ChQ4MeR6uDhtMjCAAAAADJKWp6/sqWLesZVml38OBBEwrd+wQza9YsMzxUl43o2rWrGS6qgVGXefj6669NMZnPP/88UR4DAAAAACSXqAl/DRo0kDRp0sjs2bNNgRaryZMnm3Pt0YvLqVOnzLm/oZ3a46fDR7V3EAAAAABuJlET/rSaZ7t27Uww69Chgxmmqev1TZ061azzlypVKnn88cd9wty1a9e8ttWsWdOcjxw5Uj777DM5dOiQOc6GDRvk/vvvlwsXLkitWrWS9LEBAAAAQGKLmvDnDmz58+eXhQsXSvny5c1cPV3jT+fvvfDCC1K9enWv/Vu0aGGGdK5fv96zrXnz5maen96md+/eZk0/PU7t2rVl3rx5kiFDBnnnnXeS4dEBAAAAQOKJqvBXvHhxWbdunTz66KNmsXadn6eBb9y4cX4Dmw4T1ZP2ClrNnDlTPvzwQ6lbt64pJKNr+xUtWlQ6d+4sa9eulSZNmiThowIAAACAxBc11T7dihUrJpMmTQpp359//tnv9rRp08qzzz5rTgAAAADgBFHV8wcAAAAAiB/CHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcICoDn/Xrl2L2LGuX78esWMBAAAAQEoTdeFv27Zt0q5dO8mcObOkT59eSpUqJcOHD5cbN26EfawffvhBmjZtKhkzZjSnatWqybhx4yIaKgEAAAAgJUgrUeTvv/+WBg0ayKlTp8zltGnTyu7du+WFF14w5x9//HHIxxo4cKC8++67nsvp0qWTTZs2yRNPPCE1atSQ2rVrJ8pjAAAAAIDkEFU9f3379jXB7/7775fDhw/LlStXZO7cuZI9e3b55JNPZO3atSEdZ8aMGSb4aW/fRx99JKdPnzbH0nDZu3dv06MIAAAAADeTqAl/R44ckfnz50v+/Pnlq6++kpiYGEmVKpXcc8898sYbb5h9JkyYENKxXnvtNXM+duxY6dOnj+TIkcNcLlu2rIwZM0ZuueWWRHwkAAAAAJD0oib8rVy50szra9OmjWTKlMnruk6dOpnz5cuXx3mc7du3y19//SXFixeXRx55xGxjjh8AAACAm13UhL+dO3ea8ypVqvhcp72B+fLl8+wTzG+//WbOW7ZsKcuWLZPq1aubYZ5ZsmSRtm3byp9//pkIrQcAAACAm6zgy7lz5+Tzzz+XX375RY4ePSpVq1aVESNGyLRp0yRXrlzSsWPHeB9X6TH8yZ07txw7dszM3Qs2Z+/EiRPmXOf5tWrVyuyfJk0auXDhgsyZM8e0e8WKFSYUBlO/fn2/2zdv3uw3oAIAAADATdPzt2/fPlMps1+/fmZ+nhZg0aUZtCCLBkAdnrl3794E3YfL5fK73b3Ug84DDOX2WvRF5wtqldCrV6+aXsPWrVvL+fPnZcCAAQlqIwAAAADc1OGvR48eJkT1799fFi1a5Nmugezhhx82wUuLtcRHzpw5zfnx48cD9ujp2n+6ZEMox9GewilTpkiJEiVM+3S9wOnTp5ugqsNBL1++HPQ4q1ev9nui1w8AAADATR3+dIinBj6tnDls2DCfEFahQgVzvm7dungdXytxKl2Lz27//v1y8uRJzz7BlCtXzpxr2LMXjtF5fyVLljQFYM6cOROvdgIAAADATR3+9uzZY3r2ypQpIxkyZPAZfqkFWVR8Q1XDhg3Nou46L89+jC+//NKcN2vWLM7j6LBU7f3TNf3sx9EAuWPHDrP+X6C5hQAAAADg6PCnQy5VbGxs0EIrefLkidfxdZimLu6uhVruvfde0wOoQ0C/+OILefPNNyV16tRm2KnVxYsXTXvc8wGVFoN5/PHH5ezZs+Y4a9asMYViVq1aZap96pw/nfsX1/BRAAAAAHBktc/y5ctL1qxZTc/ZwYMHfXr+li5das7r1q0b7/vQojG6lp8eq1q1al7XDR482Ge+3V133WXm7+lQ09q1a3st8r548WJzHHvVzsKFC8vIkSPj3UYAAAAAuKl7/rSn7IknnpDr169Lt27d5N9///Vcp5U/J0yYYIqpdO3aNd73UaRIEVm/fr307NnTLNKuvYj16tWTyZMnyxtvvOGzv87p03l8upSDlbZDl3MYNGiQmYuoQzx1vuAzzzwjGzZskKJFi8a7jQAAAACQEqVyBVo7IR4uXbok7dq1k4ULF/pcp/Povv32WzOk8mbm7knUyp8pxbgvpyV3EwAko55dO/H8AwlUeMARnkPAwQ4Mj5GbQUQXedeAN2/ePLNkwsyZM80aerqtVq1aplctlGqcAAAAAIAUHP60+Iou9ZA3b17p3LmzOQEAAAAAbrI5f1roRQPf66+/HqlDAgAAAABSWvjTKpnutfIAAAAAAClLxMKfVsjUhdi1B3Dv3r2ROiwAAAAAICWFPzV16lSpXLmyWYxdF03X6p8AAAAAgJuo4MvKlSuladOm5mdd6097AXWh99SpvfNlo0aNPAu+AwAAAACiLPzlyJFDmjVrFud+VatWjdRdAgAAAACSOvxVqVJFFi9eHKnDAQAAAABS6pw/AAAAAIDDwt/Vq1fl0KFDcurUqcS6CwAAAABAcoW/jRs3yj333CPZsmWTQoUKSe7cuc0yEEOHDjWBEAAAAAAQxXP+1Jo1a+S2226TixcvSkxMjFSrVs30/P32228yePBgc/3s2bN9KoACAAAAABJXRFPYU089ZYJf165dZc+ePbJgwQJZu3atOeXKlUt+/PFHmTFjRiTvEgAAAACQlOFv//798vvvv0vWrFnlk08+kYwZM3quq1mzpgwaNMj8rD1/AAAAAIAoDX8HDx4052XKlJHMmTP7XK9DQNWBAwcidZcAAAAAgKQOf9mzZzfnhw8fDhoOdTF4AAAAAECUhr9y5cpJ/vz5TfibMGGC13U6D/D99983Pzdq1ChSdwkAAAAASOpqn1rBc8iQIaboS/fu3WXhwoVSt25dU+3zq6++kl27dkmxYsWkZ8+ekbpLAAAAAEByLPXw5JNPyqVLl+S1116T6dOnm5NbvXr1ZPLkyZ7hoQAAAACAKA1/qm/fvtKjRw9ZuXKlmeenVT9vueUWqVKlSqTvCgAAAACQXOFPZcuWTe68887EODQAAAAAILkXede5fdWrV5cRI0Z4bT937pzUr19fmjRpIpcvX47kXQIAAAAAkjL8uVwueemll2TTpk3y4IMP+vQE1qpVS5YvX+5TCRQAAAAAEEXhb+/evbJv3z4pUaKEFC1a1Of6xo0bm/MVK1ZE6i4BAAAAAEkd/o4dO+bp5fPHvf3QoUORuksAAAAAQFKHv8KFC5tzXc/P37y+v/76y5wXLFgwUncJAAAAAEjq8FeoUCGpWrWqxMbGynvvved13YkTJ+TDDz80P991112RuksAAAAAQHIs9fD2229LmzZtZPDgwbJ+/XpT3VOD35dffikHDhyQ2rVrS8eOHSN5lwAAAACApA5/99xzj1nu4bnnnpMffvjBnKzXTZw4UdKmTZSlBQEAAAAAQUQ8iXXu3Fnuu+8+Wblypan+mSlTJrPMQ9myZSN9VwAAAACAECVKN5wGvpYtWybGoQEAAAAA8ZCoYzCXLFkiq1evNgvA33nnnaYHEAAAAAAQZdU+f/75ZxPqhg8f7rVdw163bt3ktttuk5dfflleeeUVqVOnjikIAwAAAACIsvD3ySefyIIFC3zm802dOtVU+KxYsaL897//lfvvv98EQg2BW7duTWibAQAAAABJFf6uXbsmP/74o6RPn17uvvtur+vGjx8vGTJkkEWLFsnAgQNlxowZ0qpVK7lx44ZMnz49vncJAAAAAEjq8KeVPC9fviwlSpSQdOnSeYVCnefXtGlTKVy4sGd7p06dzPmmTZvie5cAAAAAgKQOf0ePHjXnWbNm9dq+bds2uXTpktSrV89ruzsInj59Or53CQAAAABI6vCXJ08ec75nzx65fv26Z/uqVavMub2y55kzZ7xuBwAAAACIgvBXunRpiYmJkZMnT8rEiRPNNg2BWuhF5wE2adLEZ5ioKl68eELbDAAAAABIqvCXKlUq6d+/v/n5iSeekIYNG0rlypVNz1+XLl0kZ86cXvsvW7bMnDdv3jy+dwkAAAAASI5F3gcMGGB6/kaOHOkZ7vnAAw+Yy1bnzp2T+fPnS/bs2eX2229PyF0CAAAAAJI6/Gnvny7c/tJLL8nevXulSJEiPj1+SoeB/vnnn5IxY0azBAQAAAAAIErCX2xsrKnsmS1bNilfvrxUqVIl4L4a+MqUKRPfuwIAAAAAJNecvz/++EPq1Kkj3bt3T2gbAAAAAAApNfwBAAAAAKIH4Q8AAAAAHIDwBwAAAAAOkKBqn+rUqVMyd+7ckPfPnTu3NGjQIKF3CwAAAABIyvC3ZcsWadOmTcj762LwK1asSOjdAgAAAACSMvwVKFBA2rVrF/L+LPkAAAAAAFEY/kqXLi2ffvppZFoDAAAAAEgUFHwBAAAAAAcg/AEAAACAAyRp+Dtz5ox8++23SXmXAAAAAICEhL9ixYrJyy+/LF27do1z35MnT8qrr74qxYsXlw8//JAnHgAAAACipeCLhr+hQ4eaZRu6dOkie/fulZiYGHnsscfkrrvu8qwBOHz4cBk9erScO3dOsmfPLu3bt49k+wEAAAAAiV3tc9asWSbM3bhxw7NNh3VOmjRJSpQoIQ8++KAcPnzYLOz++uuvy7PPPis5c+ZMyF0CAAAAAJI6/L344osm+Oki7xrs9uzZI/379zfDQWNjY+X8+fPm50GDBknWrFkTclcAAAAAgOQIf/v375d//vlHsmTJIlOmTJFs2bKZ7drTN3jwYE/PYDgLwAMAAAAAUljBl4MHD5rzsmXLeoKfql27tjkvVaoUwQ8AAAAAoj38XblyxZxrz5+Ve3hnwYIFE9o2AAAAAECEsMg7AAAAADhAggq+qC1btkjr1q09l3V5B3/b3SpXrizvvPNOQu8WAAAAAJCU4U/D3o8//hjy9tOnTyf0LgEAAAAASRX+atasKX/++WfYt7PPEQQAAAAApODwlzlzZqlSpUpkWwMAAAAASBQUfAEAAAAAB4h3+Nu5c6f07t1bhg8f7jPXb+7cubJq1Sqv7TpEtFGjRvLUU0/Fv7UAAAAAgKQNf4cOHZLPPvtMZs2a5bX9r7/+kjZt2siLL77otf3MmTOycuVK2bRpU3zvEgAAAAAQTwz7BAAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcIB4L/LutmXLFmndurXXUg/BtgMAAAAAojD8aaj78ccfQ94OAAAAAIii8FezZk2zcHu4smTJEt+7BAAAAAAkdfjLnDmzVKlSJb43BwAAAAAkIQq+AAAAAIADJHjOn3K5XHL+/HnJmjWr1/aFCxeaeX96XbNmzaRz586SOjV5EwAAAACSWoKS2LVr16R///6SI0cOyZYtm+TOnVtef/11c52et2rVSkaNGiXjx4+XRx99VDp16hSpdsv169dNqIyEGzduyPHjx83p6tWrETkmAAAAANw04e+tt96SkSNHSmxsrJQpU8b0AA4ZMkTeffddeeONN6ROnTryxRdfyODBgyVdunTyzTffyIIFCxLUYC0yo6EyY8aMpqexUKFCMnToUBNE42v48OGSL18+c1q2bFmC2gcAAAAAN9WwT+150+Cnpk+fLg8++KBcuHBB7rjjDnnllVckTZo0MmfOHImJiTH76HUjRoyQWbNmmfAWH7p2YKNGjeTs2bNm+KhWDj106JAJl3v37pVx48aFfczdu3ebXsqcOXPK6dOn49UuAAAAALhpe/727dsnZ86ckaJFi5rg564A+uSTT5qhk9oT6A5+qnHjxuZ8165d8W5s3759TfB76KGHzBBN7XFcvHixCW6ff/65rF69OuxjanurVq0qDzzwQLzbBQAAAAA3bfjTHjdVuHBhr+1FihQx5zr/zypPnjyeHsD43t+iRYukQIECMnHiRMmVK5fZ3qJFCzPsU02YMCGsY06ZMkV++eUXExwpRAMAAADgZpY6IcM+lQ7vtLJfttN5gfGxatUqc9s2bdpIhgwZvK5z9zyuXLky5OOdPHnS9CQOGjSI9QoBAAAA3PSiZt0F93DRSpUq+VynhVry588vO3fuDPl4WqVUeyNffvnliLYTAAAAAG7Kdf60CEvr1q09l0+dOhV0e3zpXD/lHu5pp9uPHj0qV65ckfTp0wc91tKlS2XSpEny66+/+vQihqp+/fp+t2/evJmeRAAAAAA3X/jTUKcLuYe6Pb5SpUrlWZPPH/f2uObuXb58WXr16iW9e/eWhg0bRqx9AAAAAHBThr+aNWuaNffCpcszxIdW9FRa5dMf3a7HTps2+EN68803TTAdMGCA17EuXbrk6WHU7Xp/wY4VqLJooB5BAAAAAIjK8KfLOlSpUkWSSrly5cz5xo0bfa77999/TaCrUaNGnMf5+uuv5dixY1KqVCm/17dv396cL1++3KwpCAAAAAA3gwQP+0wqOkRTe+J04Xit1GldSuKLL74w582bN4/zODo30L3shNX58+dN71/27NklXbp05gQAAAAAN4uoqfapoU2XdNBhmW3btpW1a9fK/v375ZNPPpG3337bLDHRo0cPr9voIvQ6hPPatWuebXo73WY/de3a1Vw/Y8YMc7lu3bpJ/hgBAAAAQJze86eGDx9uhmPqen72cKZz+SpWrOi1rV27drJs2TJZt26d1K5dO4lbCwAAAAApR1SFv4IFC8qGDRtM0Fu4cKHExsZK2bJlpU+fPtKhQwef/XPkyGGGeIYyhDNr1qxm37iWiQAAAACAaJTK5XK5krsRNxN3tc9A1UCTw7gvpyV3EwAko55dO/H8AwlUeMARnkPAwQ4Mj5GbQdTM+QMAAAAAxB/hDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOEDUhr+rV6/KmTNnEnwcPcalS5ci0iYAAAAASKmiLvz98ccf0qJFC8mYMaPkzJlTYmJiZMiQISYMhuL8+fMyceJEueOOO8zt9ZQ5c2apWrWqjB07VlwuV6I/BgAAAABIamklimzevFmaNGki586dk7Rp00rWrFnl6NGj8vrrr8vevXtlwoQJcR5jyZIl8thjj3kuZ8+e3RxPj92rVy/ZunWrvP/++4n8SAAAAAAgaUVVz1+/fv1MUHv44Yfl+PHjZsjm0qVLJVeuXKY3b+XKlXEeQwNjt27dZNGiRXLq1ClzjNjYWHnjjTfM9R9++KHs2bMnCR4NAAAAACSdqAl/Bw8elMWLF0vBggVl/PjxkiNHDrO9adOmMmzYMPOzBsC4NGvWzPQQtmzZ0gz5VDrsc/DgwXLnnXeaYZ/aCwgAAAAAN5OoCX+rVq0ywaxNmzaSIUMGr+seeOABcx5Kz18wRYsWNef58uVL0HEAAAAAIKWJmvC3a9cuc16pUiWf6/LmzSv58+eXnTt3xvv4p0+flh9++EEqVKggderUSVBbAQAAACCliZqCLzrXT7mHatrpvD8t/nLlyhVJnz59WMe+ceOGdO3a1cwB/P777yV16rgzcf369f1u1yGjVapUCev+AQAAACCxRU3PnzuQaVDzx709lOBmde3aNenSpYv8+OOPMnnyZGnQoEEEWgsAAAAAKUvU9Py5e/y0yqc/uj1LlixmCYhQ6eLuHTt2lJ9++kmmTJlifg7V6tWrw+oRBAAAAIDkFDXhr1y5cp5F3u10jT8dslmzZs2Qj3f27Flp27atKRIzbdo06dChQ0TbCwAAAAApSdQM+2zYsKGkS5dO5syZIydOnPC6Tpd+UM2bNw/pWDo3UJd80Aqi3377LcEPAAAAwE0vdTQN+9RhmVr4pXXr1ia46WLso0aNkv/+97+SJk0a6dGjh9dtTp48KYcPHzbz+twOHDggTZo0kY0bN8onn3wi9erVM/tYTxcvXkyGRwgAAAAAiSdqhn2q9957T5YvXy5r1qwxPYFWb7/9tlmmwer++++XZcuWybp166R27dpm27x582T79u3m5549e/q9nzFjxkjv3r0T7XEAAAAAQFKLqvBXoEAB2bBhgwwbNkwWLlwosbGxUrZsWenTp4/ce++9Pvvnzp1bYmJizHBRt8yZM5ttweg+AAAAAHAziarwp/LkySMjR44Mad+ZM2f6bOvcubM5AQAAAICTRM2cPwAAAABA/BH+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAA0Rt+Dt//rwcOXJEbty4kSKOAwAAAAApWdSFv3Xr1knDhg0la9asUqBAAcmTJ48MGjRIrly5kizHAQAAAIBokFaiyMaNG6VZs2Zy4cIFyZgxo2TPnl2OHj0q77zzjuzbt0+mTJmSpMcBAAAAgGgRVT1/ffv2NYHt8ccfl+PHj5vhmmvWrJG8efPK1KlTZdmyZUl6HAAAAACIFlET/vbv3y9LliyRwoULy6effipZsmQx2+vWrStvvfWW+XnSpElJdhwAAAAAiCZRE/5Wr15tztu0aSPp0qXzuq59+/bmfNWqVUl2HAAAAACIJlEz52/37t3mvGLFij7X5c6dW2JiYmTXrl1Jdpz69ev73b5+/XozjzDQ9cnh6LHjyd0EAMnoi09H8fwDCXRs71WeQ8DB6q/07jRKblWqVJFx48bdvOHv3Llz5jxHjhx+r8+ZM6eZu3f58mXJkCFDoh8nkLRp00rmzJklJcmfL29yNwHJZPPmzZ43CABA/NUqnrL+8UPS4vMUN4uoCX+pU///EaqB1uO7fv26OU+TJk2SHMc9fBRIydw90Py+AgDA5ykQNXP+cuXKZc6PHTvm93qt2qlr9mnPW1IcBwAAAACiSdSEv3Llypnz33//3e88vtOnT0v58uWT7DgAAAAAEE2iJvw1atRI0qdPL3PnzjULsluNHTvWnLdo0SLJjgMAAAAA0SRqwl/27Nmlc+fOEhsbK3fffbf88ssvsm3bNnn33Xdl+PDhZtmGnj17et3m8OHDsmfPHrly5UqCjgMAAAAA0S6Vy+VySZTQeXoNGjSQHTt2+Fw3cuRI6du3r9e2Zs2aybJly2TdunVSu3bteB8HiFYUfAEAgM9TICrDn9I5edpLt3DhQtN7V7ZsWenTp4/ceeedPvs+9NBDsmbNGpkzZ45UrVo13scBAAAAgGgXdeEPAAAAAHATz/kDAAAAAMQf4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwApyoEDByRv3ryybt06c/no0aPSuHFjGTt2bHI3DQAAIKoR/gCkKIULF5ahQ4fKfffdJxkyZJDSpUtLmTJlpGvXrsndNAAAgKiWyuVyuZK7EQDgz9WrVyVdunQ8OQAAABFA+AMAAAAAB0ib3A0AAAAAwrVr1y45e/Zs0H102kDWrFn9zi8/c+aMFCxYUHLlyhXnfen9HDx4UPLlyyd58uTxu8+WLVvkypUrfq+rVKmSpE+f3mf7qVOn5NChQ5I9e3YpUqRInO0AEoo5f0AA+sHQr18/qVq1qnmjr1KlivTp00f2799vri9QoID5QInr5Hbu3DkZM2aMtGjRQooWLSr58+eXevXqyciRI83wRrsmTZpIiRIl5PLly/Kf//xHypYtaz50mjVrJgsXLvTaN9y2qN9++00eeughKV68uHl81atXlzfeeEPOnz/vty3W4+j++ry88MILcvLkSb/tjovO5dN97bfV7XZffvml5771wxcAgKeeekpq1KgR9LRmzRqvJ2rq1Knmc0aDVuXKlc3nmRYV+/333/0+oStWrDDXa0CsWLGiKUhWoUIF+e6773z2veOOOwK2w/7Zpcdt0KCBuX9th/5foO2aOXMmLywSl875A+Bt3bp1rty5c+t8WJ9Ts2bNzD5ZsmTxe7395NajR4+A+9x///0+L0GtWrVcefLkcbVt29Zn/9SpU7umT5/u2Tfctvz444+u9OnT+92nZs2artjYWJ+2BDqm+/mwtzsuMTExZl/7bXW71bFjx8zx3Pe3b98+fl0BAK5WrVqZz4Vq1ar5nAoVKmSuW7RokeeZGjdunOezJGvWrK4yZcq40qVLZy7r5+jmzZu9ntW5c+e60qZNa65PkyaNq2TJkq58+fKZy5UrV/Z5BfQ+M2bM6NWOvHnzmv13797t2W/BggWe42o7ypcvbz779HKqVKlcM2fO5NVFoqHnD7C5du2a6RHTHi39Fm/JkiVmSMZff/0lH3/8sWdYxpEjR0xvnvuk3wiWKlXKa5ue3HLkyCEvvviirFy5Ug4fPmxuv2rVKrnzzjvNN32bNm3yeS1OnDhhvrX86quvTI+jDinp37+/3LhxQ5588klPL104bbl06ZI8/vjjZmhKt27dZOPGjebxzZkzxwyP0R7BYcOG+bRFv510H+f48ePmeSlWrJgsW7bMb89lpAwYMMA8zpYtWybafQAAotcff/zhc+rbt6/XPhcvXpSBAwean4cPH26GfP7zzz/m80w/8/Vzxn290s+1Xr16mf8Jnn/+efM/gQ4z1eWHtm/fLg888IBPO/RzVXsFre3o2LGj1z56vCeeeELSpk1rRrXosM9t27aZ/wsWL15shn/qqBogsTDnD7DRMLNz505p2LCh/PTTT5ImTRrP0Eods6/DTFSWLFm8bpc6dWpJlSqV37kF6u2335bPPvtMBg0aZD5wYmNjtStOrl+/bq7XD4lbbrnF53bjx4+X1q1be5ZB0A8tHZI6ffp0WbRokdx7771htWXp0qUmLN5+++0yYcIEz3a9Dw1/Opxz2rRp8tZbb/nc1n08PY+JiTFLMej+iVWRU9uqH4669MO+ffvMByMAAOHSL1I1wLVt29Z8ieqmYeuLL74wX2jqZ6oGOJ2bp/vrZ61+Vr7//vtexypXrpy89tprPvehn+v+5vVZrV27Vvbu3SsdOnSQmjVrytatW812/X9Ap3bcfffd5jN49+7dUrJkSV5oRBw9f4CN9vAp/bbOHfwiQY/3zDPPyPLly803fPohod80ak+cunDhgs9t9JvBVq1a+Wxv06aNOf/777/Dbod+Y6l0HT07/cZST3v27PHpzdNeSPe8O/1w0yCsYVQDsp11X50nofvqOn3+ejcD0Q/g3r17m15M7TEFACC+9AtEpV/s2mXKlMkEMf3c0S9HlX4OKp2nHwotCKOf54G+AHbT3kOlcwb1y1P90ldP1apVMycNfsrdDiDS6PkDAv1xpI3cn8eGDRvk+++/N0VbtMCLvtHrMFANl9988410797d7+20B09PgdqWkGU6AwVb97F1aKmdvRiM9szp0NG5c+eaXsBA+54+fdp8u6mPVYeVaqCLi/aUarjVnljW+gMAJIT7M8o6HcPKXTXUvZ/7XD+/wvniuFChQkH3c3+e6ZenWjwmkIwZM4Z0v0C46PkDbDSgqdmzZ0fsuXH30Ok4fh1eqXPlNPzpN4Ra8SsQ/RZSewrtdGiK0nl94XIPI1mwYIHPdTqvUEOafijZw5x1zp9+GOr8w4cfftgMxbQOHw22r34r+u2334b0fGn407mJWmUNAICE0Iqa6uuvvzZVtK10zt26detMFW49Ka3Q6a4O6i8AukftuLmrf2rvXTDaw6huvfVWv3MV9aS1AbQCN5AYCH+AzW233Wa+uZs/f77p1dJhkjovTydla2GW5557LuznTEs4qylTppix/kqHfr7yyis+wcnusccek19++cUMw9QPIJ3zp/MTdJ6fzkWIz+PLmTOneSx6/8eOHTOPT3sndSiofii2b9/e723dQzk1uOpz5F7ryD2cJpR9dbJ7XLSYTbZs2eTdd98N+/EBAGCnyzXVqVPHzLlv2rSp+SJSi67pXHwd2qlftuoXjtYvgnXaxb///mtu9+mnn5ovY+fNmycvvfSSZziojmZ59tln5YMPPjBz7du1axf0ydfj6rw+HQ2kBd+0oJt+CazHnjRpknTp0sXv0FQgYhKvkCgQvRYuXGjKNbtLQmuJZ/fPTZs29XsbLftcunRpv9ddvnzZXO8+hrvEs546dOhgzseMGeOz7EGuXLlcjRs39mmDnj7++OOA7Q/WFjV58mRTTtrf49PbnThxwqct7lLYerI+N7rsxOrVq0PaV+9nw4YNQZd6cO/75Zdfel3Xq1cvlnoAAPgs9eDPe++957PUw9atW10FCxb0u2xRkyZNXBcvXvQ6xqFDh7w+u62n2rVrm31atGjh2TZkyBCfdvTp08dnqYfjx4+7br311oBLKNk/G4FIoucP8EN71P73v/+Zalzac6U9Y1qFS0s7jx49OuznTAuk6Lw4rTKmvWE6n06HoOjEbh0OGYjO99Php48++qhn/H/58uXNt4PuqqPx8cgjj5ieTV1UXdumj0+HumiBFa1wljt3br+303l87iI12nuoS2HoEFRdrD7Ufd1DXoJp3ry5+fYTAIBAdFH0QMMs9TNNr9NRJG5a0EwLj73++uum50579PRzedy4cWYKg32enVb5Xr9+vYwZM8b06OlQTd3/nXfeMSNylM7h16Ubfv31V78VQHV5KG2HtQqojoTRoZ06GqhTp05St25d0xupo410mKn2SAKJJZUmwEQ7OnCT0CGXcRUd0TWE9M8pc+bMcR5Pw5a74Ir+rLfVDx1rkZnatWubamO6BlE47Qi3LbqfHjdYeWo9nntJCqVtsM8JDHdfrW6qQ2S0ypr9trq//XHqcFRtpw531dsBAAAgPFT7BEIQSuCyhphwKm3qz3GVhg6nHeG2RYNUXOsShXO8UPf1F0yD3VYDYaDACQAAgLgx7BMAAAAAHIDwBwAAAAAOwJw/IIXS+W9aGEbnuAEAAAAJRfgDAAAAAAdg2CcAAAAAOADhDwAAAAAcgPAHAAAAAA7AOn8AAACISi6XS2bPni0//PCDbN++XWJjY6VAgQJSrFgxad26tdx9990hr5FrN3ToUPnqq6+C7jNr1iypUKFCPFsPJD3CHwAAAKLOvn37pH379rJu3Tqv7Zs2bTLnn3/+uTRq1EiWL18er+MfPnzYBMpgLl26FK9jA8mF8AcAAICocvr0aWnevLns3LlTChUqJH369DFBL2fOnHLkyBETDOfOnWt+TqgxY8ZIs2bN/F5XsmTJBB8fSEqEPwAAAESV1157zQS/WrVqycKFCyV37tw++zz++OMR6ZkrUqQIQztx0yD8AYlg48aN8v777wfdp2bNmvLss8967f/oo49K3bp1zRwDHbaSLVs2adeunTRo0MDvMf755x+ZMWOG7NmzRzJmzCh16tSRDh06SIYMGQK256mnnpJbb73V63r9cHz66afl2rVr5pvUrl27hv045s+fL9OnT4/zubn33nvNCQCA+Lhy5YqMHz9eUqVKJV9++aXf4Oemn41W2hM4atQo+fXXX+XMmTOm17BNmzbSs2dPSZ8+fbxfkF69epnhpUuXLpX8+fP7XN+jRw9ZvXq12cfd3sRqCxCUC0DEzZkzx6V/XsFO7dq189l/yJAhrooVK/rsO3DgQJ/7GDFihCtNmjQ++5YuXdq1ffv2gO255557fI41duxYz/W9evWK1+N477334txXT6+99lqEn20AgJOsXr3afJ5Ur149rNv9+eefrvz58/v9bKpfv77r/PnzXvv36dPHXKefhXGZMmWK2Vc/C+0OHjzoSps2revuu++Od1uASKHnD0hEjzzyiLRo0cJr29WrV+WJJ57wu//w4cMla9as8sorr0hMTIxs2LBBJk2aJO+88440bNjQfCOofv75Z+nfv7/5+YEHHjDzHHT+g/YYam/g/fffb3rt0qRJ43X8pk2byk8//WQmsJcvX95TKe2DDz4w8xn0G8v4Po677rpL8ubN67m8fv16+fjjj80+9evX92yvXr16yM8fAAB2+/fvN+fhVtnUUS1Hjx41n6cvvviiFC5c2Iyy0SGk2iun5++9957P7Z588kkZMGCAz3atKur+3NTCM88884zpkbTvq72TOrKme/fuCW4LkGARi5EAfHrMRo8e7fOsXLx4MWDPX968eV1Hjx712n/y5MnmujvuuMOzrU2bNmbbyJEjfY59yy23mOvmzZvnc/wPPvjAVbJkSa/evZ9++slcN3v27IA9f6E+Dqtvv/3WXK/tBwAgUqZOnWo+Xx599NGQb7Np0yZzmwoVKrguX77sdZ2OlsmQIYP5DL5x44ZPz1+gU+HChb2O88wzz5jtK1as8NpetmxZ08t35cqVeLcFiBQWeQdSEJ3zly9fPp9eN/120VrKWnvVMmfOLM8995zP3Ibnn3/e/Gwvfa20J1Bvo72JJ06cMNtGjhwpt99+u1StWjWRHhUAAJHj/pw8cOBAyLf566+/PKNl7PPpypUrZ+bMHz9+3PTG+av2uXXrVp/TsmXLvPbTuXpKe//cdB8dkdOlSxfPeoMJaQuQUIQ/IAXRkOdPwYIFzcK1bufOnTPDQlOn9v0T1gnj7n380WEn+mGjH2Z//vmnLF68WPr16xexxwAAQGLSQmNa7GXNmjVmykModKqC0qkV/ri3azGZQNU+7afSpUt77adfompw++abbzyfwe4gaB3ymZC2AAlF+ANSEP0m0e7y5cumnLW1R1AriekaRmfPnvXZ3/2Nor9qY+4PFf12Uufj/fe//5VKlSpJq1atIvo4AABILFotUz+3Lly4YObIh6Jo0aLm3N+C73qc3377zVTK1i9WE0Krep4/f16+/vprU8Hzu+++M/PyrfMTk6otgD+EPyAFmTJliin57Hb9+nUzEVxDni7B4KY/6+RxXZ7B+s2gFnLR4jDufQLRpRl0SMnUqVNNr59+gwoAQLR4++23TUDSLzJ1esS2bdu8rj958qRMnjzZUxytXr16piiZLvyuX3zqZ6h7P3fxFQ2UCV1ioVOnTpIlSxbT4zdt2jS5ePGiV69fUrYF8IfwB6Qg+m2ghrbGjRubuQD6TaGuAaQfAAMHDvTsp4FQP1z0g61MmTKmyljLli2lWrVq5kND19HThW+D3c+sWbNkwoQJ5kMTAIBoopWj9QvMTJkymS9OK1asaAKVDsXUES558uQx8+zc8/J0TvyIESPMz//5z3/MPsWKFTOjarR3TtfVdX956q/ap79hn3qaM2eO1756nAcffNAMSX3rrbc8l60S0hYgoVjqAUhBtEy0FmrRDzTr8BYNadaCLDohfN68eeYbwt27d5shoEp78Dp37iyfffZZnPd1zz33JNKjAAAg8emyRn/88YcJWdqLpoXM3MXMNAi2bt1aHn/8cc/+Ggb1i9PBgwebaRb62amF0HQpI13yKNDSEe6lJfzRoZ3+hn7q57YeX5c70gJtdvFtC5BQqbTkZ4KPAsDng0ILqejQDvsbuA7l1B47/ZbvtttuM9v0Q0vX8Bs9erQZyrllyxZTjEW//WvSpEnASeF6rLVr18qePXvMN4m1a9f2zCUItT1uWlBGv3HU9f/c6/KF+zis9u7dK0uWLDHtL1WqFL8hAIBEo//OHjt2zMy30+Jp2iMYjE590CkVOj8+0GfskSNH5NSpU0GPo0XWsmfP7rNdp2FomwJdH25bgEgh/AEpgD38AQAAAJHGnD8AAAAAcADCHwAAAAA4AMM+gRQglDl5AAAAQEIQ/gAAAADAARj2CQAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAAMjN7/8BjZ+LHT98Cu8AAAAASUVORK5CYII=",
+ "text/plain": [
+ ""
+ ]
+ },
+ "metadata": {}
+ }
+ ],
+ "source": [
+ "GREY, BLUE = \"#9aa0a6\", \"#1a73e8\"\n",
+ "fig, ax = plt.subplots(figsize=(6.5, 4.4))\n",
+ "bars = ax.bar([\"\u0441\u0442\u0430\u0440\u0442\u043e\u0432\u044b\u0439\\n\u043f\u0440\u043e\u043c\u043f\u0442\", \"\u043f\u043e\u0441\u043b\u0435\\nCoEvo\"], [init_score, final_score],\n",
+ " color=[GREY, BLUE], width=0.55)\n",
+ "for b, v in zip(bars, [init_score, final_score]):\n",
+ " ax.text(b.get_x()+b.get_width()/2, v+0.008, f\"{v:.3f}\", ha=\"center\", va=\"bottom\",\n",
+ " fontsize=12, fontweight=\"bold\")\n",
+ "ax.set_ylim(0, max(init_score, final_score)+0.1)\n",
+ "ax.set_ylabel(\"BERTScore\")\n",
+ "ax.set_title(\"CoEvo: \u0431\u044b\u043b\u043e / \u0441\u0442\u0430\u043b\u043e (SQuAD v2)\", fontsize=12, fontweight=\"bold\")\n",
+ "ax.spines[[\"top\", \"right\"]].set_visible(False)\n",
+ "plt.tight_layout()\n",
+ "plt.savefig(\"coevo_demo_result.png\", dpi=140, bbox_inches=\"tight\")\n",
+ "plt.show()"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "2ec4019e",
+ "metadata": {},
+ "source": [
+ "## \u0412\u044b\u0432\u043e\u0434\n",
+ "\n",
+ "\u0418\u0437 \u043e\u0434\u043d\u043e\u0433\u043e \u043f\u0440\u043e\u0441\u0442\u043e\u0433\u043e \u043f\u0440\u043e\u043c\u043f\u0442\u0430 CoEvo \u0441\u043e\u0431\u0440\u0430\u043b \u0441\u0442\u0440\u0443\u043a\u0442\u0443\u0440\u0438\u0440\u043e\u0432\u0430\u043d\u043d\u044b\u0439 \u043f\u0440\u043e\u043c\u043f\u0442 \u0438\u0437 \u0442\u0440\u0451\u0445 \u043f\u043e\u043b\u0435\u0439 \u0438 \u043f\u043e\u0434\u043d\u044f\u043b\n",
+ "BERTScore. \u0420\u0430\u0437\u0431\u0438\u0435\u043d\u0438\u0435 \u043d\u0430 \u0440\u043e\u043b\u044c / \u0437\u0430\u0434\u0430\u0447\u0443 / \u043e\u0433\u0440\u0430\u043d\u0438\u0447\u0435\u043d\u0438\u044f \u043f\u043e\u0437\u0432\u043e\u043b\u044f\u0435\u0442 \u0443\u043b\u0443\u0447\u0448\u0430\u0442\u044c \u043a\u043e\u043c\u043f\u043e\u043d\u0435\u043d\u0442\u044b \u043d\u0435\u0437\u0430\u0432\u0438\u0441\u0438\u043c\u043e,\n",
+ "\u0430 \u0440\u0435\u0444\u043b\u0435\u043a\u0441\u0438\u044f \u043d\u0430\u043f\u0440\u0430\u0432\u043b\u044f\u0435\u0442 \u043f\u043e\u0438\u0441\u043a.\n",
+ "\n",
+ "\u041f\u043e\u043b\u043d\u043e\u0435 \u0441\u0440\u0430\u0432\u043d\u0435\u043d\u0438\u0435 \u0441 \u0434\u0440\u0443\u0433\u0438\u043c\u0438 \u043c\u0435\u0442\u043e\u0434\u0430\u043c\u0438 \u043d\u0430 \u0448\u0435\u0441\u0442\u0438 \u0434\u0430\u0442\u0430\u0441\u0435\u0442\u0430\u0445 \u2014 \u0432 `benchmark_coevo.png`."
+ ]
+ }
+ ],
+ "metadata": {
+ "kernelspec": {
+ "display_name": "CoolPrompt (.venv)",
+ "language": "python",
+ "name": "coolprompt-demo"
+ },
+ "language_info": {
+ "codemirror_mode": {
+ "name": "ipython",
+ "version": 3
+ },
+ "file_extension": ".py",
+ "mimetype": "text/x-python",
+ "name": "python",
+ "nbconvert_exporter": "python",
+ "pygments_lexer": "ipython3",
+ "version": "3.12.13"
+ },
+ "widgets": {
+ "application/vnd.jupyter.widget-state+json": {
+ "state": {
+ "031e604e237b4c0a95d74c6b527ed3cb": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "FloatProgressModel",
+ "state": {
+ "_dom_classes": [],
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "FloatProgressModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/controls",
+ "_view_module_version": "2.0.0",
+ "_view_name": "ProgressView",
+ "bar_style": "success",
+ "description": "",
+ "description_allow_html": false,
+ "layout": "IPY_MODEL_584f4315bb714773a440f64e9cd909bd",
+ "max": 103.0,
+ "min": 0.0,
+ "orientation": "horizontal",
+ "style": "IPY_MODEL_fb6ea2a20a9a4b55875f918ae26999f0",
+ "tabbable": null,
+ "tooltip": null,
+ "value": 103.0
+ }
+ },
+ "077570ad5c0d48b785739da1415a35b4": {
+ "model_module": "@jupyter-widgets/base",
+ "model_module_version": "2.0.0",
+ "model_name": "LayoutModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/base",
+ "_model_module_version": "2.0.0",
+ "_model_name": "LayoutModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "LayoutView",
+ "align_content": null,
+ "align_items": null,
+ "align_self": null,
+ "border_bottom": null,
+ "border_left": null,
+ "border_right": null,
+ "border_top": null,
+ "bottom": null,
+ "display": null,
+ "flex": null,
+ "flex_flow": null,
+ "grid_area": null,
+ "grid_auto_columns": null,
+ "grid_auto_flow": null,
+ "grid_auto_rows": null,
+ "grid_column": null,
+ "grid_gap": null,
+ "grid_row": null,
+ "grid_template_areas": null,
+ "grid_template_columns": null,
+ "grid_template_rows": null,
+ "height": null,
+ "justify_content": null,
+ "justify_items": null,
+ "left": null,
+ "margin": null,
+ "max_height": null,
+ "max_width": null,
+ "min_height": null,
+ "min_width": null,
+ "object_fit": null,
+ "object_position": null,
+ "order": null,
+ "overflow": null,
+ "padding": null,
+ "right": null,
+ "top": null,
+ "visibility": null,
+ "width": null
+ }
+ },
+ "08c8247e7de641859d6d192e53d3c508": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "HBoxModel",
+ "state": {
+ "_dom_classes": [],
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "HBoxModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/controls",
+ "_view_module_version": "2.0.0",
+ "_view_name": "HBoxView",
+ "box_style": "",
+ "children": [
+ "IPY_MODEL_2a49562fea2e472bb16c15aae1f8d6c1",
+ "IPY_MODEL_031e604e237b4c0a95d74c6b527ed3cb",
+ "IPY_MODEL_82716763d58d4919a89d197bc7a08987"
+ ],
+ "layout": "IPY_MODEL_195a14f673404c12ac8918eb5b58f578",
+ "tabbable": null,
+ "tooltip": null
+ }
+ },
+ "195a14f673404c12ac8918eb5b58f578": {
+ "model_module": "@jupyter-widgets/base",
+ "model_module_version": "2.0.0",
+ "model_name": "LayoutModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/base",
+ "_model_module_version": "2.0.0",
+ "_model_name": "LayoutModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "LayoutView",
+ "align_content": null,
+ "align_items": null,
+ "align_self": null,
+ "border_bottom": null,
+ "border_left": null,
+ "border_right": null,
+ "border_top": null,
+ "bottom": null,
+ "display": null,
+ "flex": null,
+ "flex_flow": null,
+ "grid_area": null,
+ "grid_auto_columns": null,
+ "grid_auto_flow": null,
+ "grid_auto_rows": null,
+ "grid_column": null,
+ "grid_gap": null,
+ "grid_row": null,
+ "grid_template_areas": null,
+ "grid_template_columns": null,
+ "grid_template_rows": null,
+ "height": null,
+ "justify_content": null,
+ "justify_items": null,
+ "left": null,
+ "margin": null,
+ "max_height": null,
+ "max_width": null,
+ "min_height": null,
+ "min_width": null,
+ "object_fit": null,
+ "object_position": null,
+ "order": null,
+ "overflow": null,
+ "padding": null,
+ "right": null,
+ "top": null,
+ "visibility": null,
+ "width": null
+ }
+ },
+ "203198d1b2f849c38bb3225abfff09ae": {
+ "model_module": "@jupyter-widgets/base",
+ "model_module_version": "2.0.0",
+ "model_name": "LayoutModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/base",
+ "_model_module_version": "2.0.0",
+ "_model_name": "LayoutModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "LayoutView",
+ "align_content": null,
+ "align_items": null,
+ "align_self": null,
+ "border_bottom": null,
+ "border_left": null,
+ "border_right": null,
+ "border_top": null,
+ "bottom": null,
+ "display": null,
+ "flex": null,
+ "flex_flow": null,
+ "grid_area": null,
+ "grid_auto_columns": null,
+ "grid_auto_flow": null,
+ "grid_auto_rows": null,
+ "grid_column": null,
+ "grid_gap": null,
+ "grid_row": null,
+ "grid_template_areas": null,
+ "grid_template_columns": null,
+ "grid_template_rows": null,
+ "height": null,
+ "justify_content": null,
+ "justify_items": null,
+ "left": null,
+ "margin": null,
+ "max_height": null,
+ "max_width": null,
+ "min_height": null,
+ "min_width": null,
+ "object_fit": null,
+ "object_position": null,
+ "order": null,
+ "overflow": null,
+ "padding": null,
+ "right": null,
+ "top": null,
+ "visibility": null,
+ "width": null
+ }
+ },
+ "24450aab261649438b0acbaa29232ba6": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "HTMLStyleModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "HTMLStyleModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "StyleView",
+ "background": null,
+ "description_width": "",
+ "font_size": null,
+ "text_color": null
+ }
+ },
+ "2a49562fea2e472bb16c15aae1f8d6c1": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "HTMLModel",
+ "state": {
+ "_dom_classes": [],
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "HTMLModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/controls",
+ "_view_module_version": "2.0.0",
+ "_view_name": "HTMLView",
+ "description": "",
+ "description_allow_html": false,
+ "layout": "IPY_MODEL_b5a24bfa925e4def98613ee86b46fad2",
+ "placeholder": "\u200b",
+ "style": "IPY_MODEL_5515709d31e74d0786c709788dab537a",
+ "tabbable": null,
+ "tooltip": null,
+ "value": "Loading\u2007weights:\u2007100%"
+ }
+ },
+ "5515709d31e74d0786c709788dab537a": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "HTMLStyleModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "HTMLStyleModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "StyleView",
+ "background": null,
+ "description_width": "",
+ "font_size": null,
+ "text_color": null
+ }
+ },
+ "584f4315bb714773a440f64e9cd909bd": {
+ "model_module": "@jupyter-widgets/base",
+ "model_module_version": "2.0.0",
+ "model_name": "LayoutModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/base",
+ "_model_module_version": "2.0.0",
+ "_model_name": "LayoutModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "LayoutView",
+ "align_content": null,
+ "align_items": null,
+ "align_self": null,
+ "border_bottom": null,
+ "border_left": null,
+ "border_right": null,
+ "border_top": null,
+ "bottom": null,
+ "display": null,
+ "flex": null,
+ "flex_flow": null,
+ "grid_area": null,
+ "grid_auto_columns": null,
+ "grid_auto_flow": null,
+ "grid_auto_rows": null,
+ "grid_column": null,
+ "grid_gap": null,
+ "grid_row": null,
+ "grid_template_areas": null,
+ "grid_template_columns": null,
+ "grid_template_rows": null,
+ "height": null,
+ "justify_content": null,
+ "justify_items": null,
+ "left": null,
+ "margin": null,
+ "max_height": null,
+ "max_width": null,
+ "min_height": null,
+ "min_width": null,
+ "object_fit": null,
+ "object_position": null,
+ "order": null,
+ "overflow": null,
+ "padding": null,
+ "right": null,
+ "top": null,
+ "visibility": null,
+ "width": null
+ }
+ },
+ "7bbbde5c6f1d4b4a8767656709b2fad5": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "HTMLModel",
+ "state": {
+ "_dom_classes": [],
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "HTMLModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/controls",
+ "_view_module_version": "2.0.0",
+ "_view_name": "HTMLView",
+ "description": "",
+ "description_allow_html": false,
+ "layout": "IPY_MODEL_f0befdf41bdd4983a71de8cc467dadff",
+ "placeholder": "\u200b",
+ "style": "IPY_MODEL_24450aab261649438b0acbaa29232ba6",
+ "tabbable": null,
+ "tooltip": null,
+ "value": "\u2007199/199\u2007[00:00<00:00,\u20076495.81it/s]"
+ }
+ },
+ "80bd5e6f939342c6b6ad6571ac8e3e20": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "ProgressStyleModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "ProgressStyleModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "StyleView",
+ "bar_color": null,
+ "description_width": ""
+ }
+ },
+ "8189b3e8af0b42b4bb9024b06c15e584": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "FloatProgressModel",
+ "state": {
+ "_dom_classes": [],
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "FloatProgressModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/controls",
+ "_view_module_version": "2.0.0",
+ "_view_name": "ProgressView",
+ "bar_style": "success",
+ "description": "",
+ "description_allow_html": false,
+ "layout": "IPY_MODEL_077570ad5c0d48b785739da1415a35b4",
+ "max": 199.0,
+ "min": 0.0,
+ "orientation": "horizontal",
+ "style": "IPY_MODEL_80bd5e6f939342c6b6ad6571ac8e3e20",
+ "tabbable": null,
+ "tooltip": null,
+ "value": 199.0
+ }
+ },
+ "82716763d58d4919a89d197bc7a08987": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "HTMLModel",
+ "state": {
+ "_dom_classes": [],
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "HTMLModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/controls",
+ "_view_module_version": "2.0.0",
+ "_view_name": "HTMLView",
+ "description": "",
+ "description_allow_html": false,
+ "layout": "IPY_MODEL_203198d1b2f849c38bb3225abfff09ae",
+ "placeholder": "\u200b",
+ "style": "IPY_MODEL_b25121ecfa3349aebd6d7c2c2a53b64b",
+ "tabbable": null,
+ "tooltip": null,
+ "value": "\u2007103/103\u2007[00:00<00:00,\u20075296.42it/s]"
+ }
+ },
+ "97ac6d20674a45988b8cbf2033cb72eb": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "HBoxModel",
+ "state": {
+ "_dom_classes": [],
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "HBoxModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/controls",
+ "_view_module_version": "2.0.0",
+ "_view_name": "HBoxView",
+ "box_style": "",
+ "children": [
+ "IPY_MODEL_a554639955504d199eeaf5dd024b60e9",
+ "IPY_MODEL_8189b3e8af0b42b4bb9024b06c15e584",
+ "IPY_MODEL_7bbbde5c6f1d4b4a8767656709b2fad5"
+ ],
+ "layout": "IPY_MODEL_9c19a5097bb14d5da57bf82ffd2562ec",
+ "tabbable": null,
+ "tooltip": null
+ }
+ },
+ "9c19a5097bb14d5da57bf82ffd2562ec": {
+ "model_module": "@jupyter-widgets/base",
+ "model_module_version": "2.0.0",
+ "model_name": "LayoutModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/base",
+ "_model_module_version": "2.0.0",
+ "_model_name": "LayoutModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "LayoutView",
+ "align_content": null,
+ "align_items": null,
+ "align_self": null,
+ "border_bottom": null,
+ "border_left": null,
+ "border_right": null,
+ "border_top": null,
+ "bottom": null,
+ "display": null,
+ "flex": null,
+ "flex_flow": null,
+ "grid_area": null,
+ "grid_auto_columns": null,
+ "grid_auto_flow": null,
+ "grid_auto_rows": null,
+ "grid_column": null,
+ "grid_gap": null,
+ "grid_row": null,
+ "grid_template_areas": null,
+ "grid_template_columns": null,
+ "grid_template_rows": null,
+ "height": null,
+ "justify_content": null,
+ "justify_items": null,
+ "left": null,
+ "margin": null,
+ "max_height": null,
+ "max_width": null,
+ "min_height": null,
+ "min_width": null,
+ "object_fit": null,
+ "object_position": null,
+ "order": null,
+ "overflow": null,
+ "padding": null,
+ "right": null,
+ "top": null,
+ "visibility": null,
+ "width": null
+ }
+ },
+ "a554639955504d199eeaf5dd024b60e9": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "HTMLModel",
+ "state": {
+ "_dom_classes": [],
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "HTMLModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/controls",
+ "_view_module_version": "2.0.0",
+ "_view_name": "HTMLView",
+ "description": "",
+ "description_allow_html": false,
+ "layout": "IPY_MODEL_ab57d75d1029443483be38ee60fec37e",
+ "placeholder": "\u200b",
+ "style": "IPY_MODEL_eb7518fe02e74cda9d5b0dfdd010256b",
+ "tabbable": null,
+ "tooltip": null,
+ "value": "Loading\u2007weights:\u2007100%"
+ }
+ },
+ "ab57d75d1029443483be38ee60fec37e": {
+ "model_module": "@jupyter-widgets/base",
+ "model_module_version": "2.0.0",
+ "model_name": "LayoutModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/base",
+ "_model_module_version": "2.0.0",
+ "_model_name": "LayoutModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "LayoutView",
+ "align_content": null,
+ "align_items": null,
+ "align_self": null,
+ "border_bottom": null,
+ "border_left": null,
+ "border_right": null,
+ "border_top": null,
+ "bottom": null,
+ "display": null,
+ "flex": null,
+ "flex_flow": null,
+ "grid_area": null,
+ "grid_auto_columns": null,
+ "grid_auto_flow": null,
+ "grid_auto_rows": null,
+ "grid_column": null,
+ "grid_gap": null,
+ "grid_row": null,
+ "grid_template_areas": null,
+ "grid_template_columns": null,
+ "grid_template_rows": null,
+ "height": null,
+ "justify_content": null,
+ "justify_items": null,
+ "left": null,
+ "margin": null,
+ "max_height": null,
+ "max_width": null,
+ "min_height": null,
+ "min_width": null,
+ "object_fit": null,
+ "object_position": null,
+ "order": null,
+ "overflow": null,
+ "padding": null,
+ "right": null,
+ "top": null,
+ "visibility": null,
+ "width": null
+ }
+ },
+ "b25121ecfa3349aebd6d7c2c2a53b64b": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "HTMLStyleModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "HTMLStyleModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "StyleView",
+ "background": null,
+ "description_width": "",
+ "font_size": null,
+ "text_color": null
+ }
+ },
+ "b5a24bfa925e4def98613ee86b46fad2": {
+ "model_module": "@jupyter-widgets/base",
+ "model_module_version": "2.0.0",
+ "model_name": "LayoutModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/base",
+ "_model_module_version": "2.0.0",
+ "_model_name": "LayoutModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "LayoutView",
+ "align_content": null,
+ "align_items": null,
+ "align_self": null,
+ "border_bottom": null,
+ "border_left": null,
+ "border_right": null,
+ "border_top": null,
+ "bottom": null,
+ "display": null,
+ "flex": null,
+ "flex_flow": null,
+ "grid_area": null,
+ "grid_auto_columns": null,
+ "grid_auto_flow": null,
+ "grid_auto_rows": null,
+ "grid_column": null,
+ "grid_gap": null,
+ "grid_row": null,
+ "grid_template_areas": null,
+ "grid_template_columns": null,
+ "grid_template_rows": null,
+ "height": null,
+ "justify_content": null,
+ "justify_items": null,
+ "left": null,
+ "margin": null,
+ "max_height": null,
+ "max_width": null,
+ "min_height": null,
+ "min_width": null,
+ "object_fit": null,
+ "object_position": null,
+ "order": null,
+ "overflow": null,
+ "padding": null,
+ "right": null,
+ "top": null,
+ "visibility": null,
+ "width": null
+ }
+ },
+ "eb7518fe02e74cda9d5b0dfdd010256b": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "HTMLStyleModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "HTMLStyleModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "StyleView",
+ "background": null,
+ "description_width": "",
+ "font_size": null,
+ "text_color": null
+ }
+ },
+ "f0befdf41bdd4983a71de8cc467dadff": {
+ "model_module": "@jupyter-widgets/base",
+ "model_module_version": "2.0.0",
+ "model_name": "LayoutModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/base",
+ "_model_module_version": "2.0.0",
+ "_model_name": "LayoutModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "LayoutView",
+ "align_content": null,
+ "align_items": null,
+ "align_self": null,
+ "border_bottom": null,
+ "border_left": null,
+ "border_right": null,
+ "border_top": null,
+ "bottom": null,
+ "display": null,
+ "flex": null,
+ "flex_flow": null,
+ "grid_area": null,
+ "grid_auto_columns": null,
+ "grid_auto_flow": null,
+ "grid_auto_rows": null,
+ "grid_column": null,
+ "grid_gap": null,
+ "grid_row": null,
+ "grid_template_areas": null,
+ "grid_template_columns": null,
+ "grid_template_rows": null,
+ "height": null,
+ "justify_content": null,
+ "justify_items": null,
+ "left": null,
+ "margin": null,
+ "max_height": null,
+ "max_width": null,
+ "min_height": null,
+ "min_width": null,
+ "object_fit": null,
+ "object_position": null,
+ "order": null,
+ "overflow": null,
+ "padding": null,
+ "right": null,
+ "top": null,
+ "visibility": null,
+ "width": null
+ }
+ },
+ "fb6ea2a20a9a4b55875f918ae26999f0": {
+ "model_module": "@jupyter-widgets/controls",
+ "model_module_version": "2.0.0",
+ "model_name": "ProgressStyleModel",
+ "state": {
+ "_model_module": "@jupyter-widgets/controls",
+ "_model_module_version": "2.0.0",
+ "_model_name": "ProgressStyleModel",
+ "_view_count": null,
+ "_view_module": "@jupyter-widgets/base",
+ "_view_module_version": "2.0.0",
+ "_view_name": "StyleView",
+ "bar_color": null,
+ "description_width": ""
+ }
+ }
+ },
+ "version_major": 2,
+ "version_minor": 0
+ }
+ }
+ },
+ "nbformat": 4,
+ "nbformat_minor": 5
+}
\ No newline at end of file
diff --git a/notebooks/examples/coevo_demo_result.png b/notebooks/examples/coevo_demo_result.png
new file mode 100644
index 00000000..0219f067
Binary files /dev/null and b/notebooks/examples/coevo_demo_result.png differ
diff --git a/notebooks/experiments/pipeline/config.example.yaml b/notebooks/experiments/pipeline/config.example.yaml
new file mode 100644
index 00000000..7c3ab6bb
--- /dev/null
+++ b/notebooks/experiments/pipeline/config.example.yaml
@@ -0,0 +1,31 @@
+openai_api_keys:
+ - "sk-..." # key 0
+# - "sk-..." # key 1 (optional, add more for higher RPM)
+# active_keys: [0, 1] # indices of keys to use; omit to use all
+openrouter_api_key: "" # only needed when provider: openrouter
+
+provider: openai
+seed: 42
+temperature: 0.7
+model: gpt-4o-mini
+population_size: 5
+num_epochs: 6
+train_size: 50
+val_size: 80
+use_enhancements: true
+use_dedup: true
+evolve_constraints: false
+factorized_phase_epochs: [4, 4, 3]
+output_dir: ../optimization_results
+requests_per_minute:
+ openai: 490
+ openrouter: 5000
+datasets_to_run:
+ - tweet_eval
+ - xsum
+ - common_gen
+ - gsm8k
+ - squad_v2
+ - mediqa
+role_modes_to_run:
+ - coevo_enhanced
diff --git a/notebooks/experiments/pipeline/evaluate_prompts.py b/notebooks/experiments/pipeline/evaluate_prompts.py
new file mode 100644
index 00000000..69b19126
--- /dev/null
+++ b/notebooks/experiments/pipeline/evaluate_prompts.py
@@ -0,0 +1,376 @@
+import os
+import sys
+import json
+import yaml
+import time
+import argparse
+import traceback
+from datetime import datetime
+
+_script_dir = os.path.dirname(os.path.abspath(__file__))
+sys.path.append(os.path.abspath(os.path.join(_script_dir, "../../../")))
+sys.path.append(os.path.abspath(os.path.join(_script_dir, "../../../src")))
+
+import torch
+import gc
+
+from langchain_core.globals import set_llm_cache
+from langchain_community.cache import SQLiteCache
+
+from coolprompt.optimizer.reflective_prompt.prompt import Prompt
+from coolprompt.evaluator import Evaluator, validate_and_create_metric
+from coolprompt.utils.logging_config import setup_logging
+
+from model_utils import create_model, normalize_model_name
+from dataset_config import DATASETS_CONFIG, load_eval_data
+
+setup_logging()
+
+_EVAL_CONFIG_PATH = os.path.join(_script_dir, "evaluation_config.yaml")
+_PROXY_CONFIG_PATH = os.path.join(_script_dir, "proxy_config.yaml")
+
+with open(_EVAL_CONFIG_PATH) as f:
+ EVAL_CONFIG = yaml.safe_load(f)
+
+PROXY_CONFIG = (
+ yaml.safe_load(open(_PROXY_CONFIG_PATH))
+ if os.path.exists(_PROXY_CONFIG_PATH)
+ else {}
+)
+
+DEFAULT_TEST_SIZE = EVAL_CONFIG.get("default_test_size", 1000)
+DEFAULT_SEED = EVAL_CONFIG.get("default_seed", 42)
+DEFAULT_PROVIDER = EVAL_CONFIG.get("provider", "openai")
+DEFAULT_MODEL = EVAL_CONFIG.get("model", "gpt-4o-mini")
+TEMPERATURE_CONFIG = EVAL_CONFIG.get("temperature", 0.0)
+
+REQUESTS_PER_MINUTE_CONFIG = EVAL_CONFIG.get(
+ "requests_per_minute", {"openai": 200, "openrouter": 5000}
+)
+
+_COMBO_MAP = {1: "text_only", 2: "text_role", 3: "text_role_constraints"}
+_raw_combo = EVAL_CONFIG.get("combo", "all")
+if str(_raw_combo).strip().lower() == "all":
+ COMBO_FILTER = None
+else:
+ _items = (
+ _raw_combo
+ if isinstance(_raw_combo, list)
+ else str(_raw_combo).split(",")
+ )
+ COMBO_FILTER = {_COMBO_MAP[int(x)] for x in _items}
+
+_raw_output_dir = EVAL_CONFIG.get("output_dir", "./evaluation_results")
+EVAL_OUTPUT_DIR = (
+ _raw_output_dir
+ if os.path.isabs(_raw_output_dir)
+ else os.path.join(_script_dir, _raw_output_dir)
+)
+
+_raw_paths = EVAL_CONFIG.get("dataset_paths", {})
+DATASET_PATHS = {
+ k: (
+ [os.path.join(_script_dir, p) if not os.path.isabs(p) else p for p in v]
+ if isinstance(v, list)
+ else (os.path.join(_script_dir, v) if not os.path.isabs(v) else v)
+ )
+ for k, v in _raw_paths.items()
+}
+
+_first_proxy = (PROXY_CONFIG.get("proxies") or [None])[0] or PROXY_CONFIG.get(
+ "proxy", {}
+).get("http")
+if _first_proxy:
+ os.environ.setdefault("HTTP_PROXY", _first_proxy)
+ os.environ.setdefault("HTTPS_PROXY", _first_proxy)
+ print(f"Proxy set for HF downloads: {_first_proxy}")
+
+
+def evaluate_single_dataset(
+ dataset_name,
+ json_path,
+ inputs,
+ targets,
+ provider,
+ model_name,
+ requests_per_minute,
+ seed=None,
+ full_test=False,
+):
+ if not json_path or not os.path.exists(json_path):
+ print(f"skip {dataset_name}: no valid json path")
+ return None
+
+ with open(json_path) as f:
+ data = json.load(f)
+
+ effective_model_name = normalize_model_name(provider, model_name)
+
+ candidates_meta = data.get("candidates")
+ if candidates_meta:
+ prompts_to_eval = [
+ {
+ "combo": c["combo"],
+ "prompt": c["prompt"],
+ "role": c.get("role", ""),
+ "constraints": c.get("constraints", ""),
+ "val_score": c.get("val_score"),
+ }
+ for c in candidates_meta
+ if c.get("prompt")
+ ]
+ else:
+ best_prompt_text = data.get("best_prompt")
+ if not best_prompt_text:
+ print(f"no best_prompt in {dataset_name} json")
+ return None
+ prompts_to_eval = [
+ {
+ "combo": "default",
+ "prompt": best_prompt_text,
+ "role": data.get("best_role") or "",
+ "constraints": data.get("best_constraints") or "",
+ "val_score": data.get("best_score"),
+ }
+ ]
+
+ if COMBO_FILTER:
+ prompts_to_eval = [
+ p for p in prompts_to_eval if p["combo"] in COMBO_FILTER
+ ]
+ print(f"{dataset_name}: {len(prompts_to_eval)} combos")
+
+ config = DATASETS_CONFIG[dataset_name]
+ model = create_model(
+ provider,
+ model_name,
+ requests_per_minute,
+ EVAL_CONFIG,
+ PROXY_CONFIG,
+ temperature=TEMPERATURE_CONFIG,
+ )
+ metric = validate_and_create_metric(config["task"], config["metric"])
+ evaluator = Evaluator(model, config["task"], metric)
+
+ candidate_results = []
+ for p in prompts_to_eval:
+ prompt_obj = Prompt(
+ text=p["prompt"], role=p["role"], constraints=p["constraints"]
+ )
+ try:
+ score = evaluator.evaluate(
+ prompt=prompt_obj.text,
+ dataset=inputs,
+ targets=targets,
+ system_role=prompt_obj.role or None,
+ constraints=prompt_obj.constraints or None,
+ )
+ print(f" [{p['combo']}] test={score:.4f} val={p['val_score']}")
+ candidate_results.append(
+ {
+ "combo": p["combo"],
+ "prompt": p["prompt"],
+ "role": p["role"],
+ "constraints": p["constraints"],
+ "val_score": p["val_score"],
+ "test_score": score,
+ }
+ )
+ except Exception as e:
+ print(f" error [{p['combo']}]: {e}")
+ traceback.print_exc()
+
+ if not candidate_results:
+ return None
+
+ best_result = max(
+ candidate_results,
+ key=lambda r: (
+ r["val_score"]
+ if r.get("val_score") is not None
+ else r["test_score"]
+ ),
+ )
+ print(
+ f"best: {best_result['combo']} = {best_result['test_score']:.4f} (val={best_result.get('val_score')})"
+ )
+
+ opt_params = data.get("parameters", {})
+ return {
+ "dataset": dataset_name,
+ "role_mode": data.get("role_mode", "unknown"),
+ "score": best_result["test_score"],
+ "best_combo": best_result["combo"],
+ "metric": config["metric"],
+ "num_samples": len(inputs),
+ "seed": seed,
+ "full_test": full_test,
+ "provider": provider,
+ "model": effective_model_name,
+ "opt_temperature": opt_params.get("temperature", 0.7),
+ "val_temperature": opt_params.get("val_temperature", 0.7),
+ "use_enhancements": opt_params.get("use_enhancements", None),
+ "json_path": json_path,
+ "candidate_results": candidate_results,
+ "prompt_info": data,
+ }
+
+
+def main():
+ parser = argparse.ArgumentParser(description="Evaluate optimized prompts")
+ parser.add_argument("--seed", type=int, default=DEFAULT_SEED)
+ parser.add_argument("--num_samples", type=int, default=DEFAULT_TEST_SIZE)
+ parser.add_argument("--dataset", type=str)
+ parser.add_argument(
+ "--provider",
+ type=str,
+ default=DEFAULT_PROVIDER,
+ choices=["openai", "openrouter"],
+ )
+ parser.add_argument("--model", type=str, default=DEFAULT_MODEL)
+ parser.add_argument("--requests_per_minute", type=int, default=None)
+ parser.add_argument(
+ "--full_test",
+ action="store_true",
+ help="Evaluate on the full split (no seed/size limit)",
+ )
+ args = parser.parse_args()
+
+ if args.requests_per_minute is None:
+ if isinstance(REQUESTS_PER_MINUTE_CONFIG, dict):
+ args.requests_per_minute = REQUESTS_PER_MINUTE_CONFIG.get(
+ args.provider, 500
+ )
+ else:
+ args.requests_per_minute = int(REQUESTS_PER_MINUTE_CONFIG)
+
+ print(
+ f"Provider: {args.provider}, Model: {args.model}, RPM: {args.requests_per_minute}"
+ )
+
+ set_llm_cache(SQLiteCache(database_path=".langchain.db"))
+
+ output_dir = (
+ EVAL_OUTPUT_DIR.rstrip("/\\") + "_full"
+ if args.full_test
+ else EVAL_OUTPUT_DIR
+ )
+
+ datasets_to_run = DATASET_PATHS
+ if args.dataset:
+ if args.dataset not in DATASET_PATHS:
+ print(f"unknown dataset: {args.dataset}")
+ return
+ datasets_to_run = {args.dataset: DATASET_PATHS[args.dataset]}
+
+ results = {}
+
+ for dataset_name, dataset_json_paths in datasets_to_run.items():
+ if not dataset_json_paths:
+ print(f"skip {dataset_name}: no json path")
+ continue
+ if dataset_name not in DATASETS_CONFIG:
+ print(f"skip {dataset_name}: not in config")
+ continue
+
+ if isinstance(dataset_json_paths, str):
+ dataset_json_paths = [dataset_json_paths]
+ elif not isinstance(dataset_json_paths, list):
+ print(f"skip {dataset_name}: bad path type")
+ continue
+
+ print(f"\n{dataset_name}")
+
+ config = DATASETS_CONFIG[dataset_name]
+ try:
+ inputs, targets = load_eval_data(
+ dataset_name,
+ config,
+ args.num_samples,
+ args.seed,
+ args.full_test,
+ )
+ except Exception as e:
+ print(f"failed to load {dataset_name}: {e}")
+ continue
+
+ if not inputs:
+ print(f"no data: {dataset_name}")
+ continue
+
+ dataset_results = []
+ for json_path in dataset_json_paths:
+ max_retries = 3
+ dataset_result = None
+ run_timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
+
+ for attempt in range(max_retries):
+ try:
+ dataset_result = evaluate_single_dataset(
+ dataset_name,
+ json_path,
+ inputs,
+ targets,
+ args.provider,
+ args.model,
+ args.requests_per_minute,
+ seed=args.seed,
+ full_test=args.full_test,
+ )
+ if dataset_result:
+ break
+ except Exception as e:
+ print(f"error (attempt {attempt + 1}): {e}")
+ if "429" in str(e) or "Rate limit" in str(e):
+ time.sleep(60 * (attempt + 1))
+ else:
+ time.sleep(10)
+
+ if dataset_result:
+ dataset_results.append(dataset_result)
+
+ role_mode = dataset_result.get("role_mode", "unknown")
+ method_dir = os.path.join(output_dir, dataset_name, role_mode)
+ os.makedirs(method_dir, exist_ok=True)
+ score = dataset_result.get("score")
+ score_str = (
+ f"{score:.2f}" if isinstance(score, (int, float)) else "NA"
+ )
+ seed_str = "all" if args.full_test else str(args.seed)
+ filename = f"{run_timestamp}_{score_str}_{role_mode}_seed{seed_str}.json"
+ save_path = os.path.join(method_dir, filename)
+ with open(save_path, "w") as f:
+ json.dump(dataset_result, f, indent=2)
+ print(f"saved: {save_path}")
+
+ if dataset_results:
+ results[dataset_name] = dataset_results
+
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ print("\nsummary:")
+ for name, dataset_runs in results.items():
+ for idx, res in enumerate(dataset_runs, start=1):
+ combo_info = (
+ f" [{res.get('best_combo', 'default')}]"
+ if res.get("best_combo")
+ else ""
+ )
+ print(
+ f" {name} [run {idx}]{combo_info}: {res['score']:.4f} ({res['metric']}) - {res['num_samples']} samples"
+ )
+ if (
+ res.get("candidate_results")
+ and len(res["candidate_results"]) > 1
+ ):
+ for cr in res["candidate_results"]:
+ vs = cr.get("val_score")
+ vs_str = f"{vs:.4f}" if isinstance(vs, float) else str(vs)
+ print(
+ f" {cr['combo']}: val={vs_str} test={cr['test_score']:.4f}"
+ )
+
+
+if __name__ == "__main__":
+ main()
diff --git a/notebooks/experiments/pipeline/evaluation_config.example.yaml b/notebooks/experiments/pipeline/evaluation_config.example.yaml
new file mode 100644
index 00000000..9ee6ded4
--- /dev/null
+++ b/notebooks/experiments/pipeline/evaluation_config.example.yaml
@@ -0,0 +1,18 @@
+openai_api_keys:
+ - "sk-..." # key 0
+# active_keys: [0] # indices of keys to use; omit to use all
+openrouter_api_key: "" # only needed when provider: openrouter
+
+provider: openai
+model: gpt-4o-mini
+temperature: 0.0
+default_test_size: 1000
+default_seed: 42
+requests_per_minute:
+ openai: 200
+ openrouter: 5000
+combo: all # all | 1 (text_only) | 2 (text_role) | 3 (text_role_constraints)
+output_dir: ./evaluation_results
+
+# Paths to optimization result JSON files, populated automatically by optmize_single.py
+dataset_paths: {}
diff --git a/notebooks/experiments/pipeline/model_utils.py b/notebooks/experiments/pipeline/model_utils.py
new file mode 100644
index 00000000..b0dcd9df
--- /dev/null
+++ b/notebooks/experiments/pipeline/model_utils.py
@@ -0,0 +1,129 @@
+import os
+import time
+import httpx
+from langchain_openai import ChatOpenAI
+from langchain_core.rate_limiters import InMemoryRateLimiter
+
+
+class MultiKeyModel:
+ _MAX_INNER_CONCURRENCY = 8
+
+ def __init__(self, models: list):
+ self._models = models
+
+ def batch(self, requests: list) -> list:
+ n = len(self._models)
+ if n == 1:
+ return self._batch_with_retry(self._models[0], requests)
+
+ chunks = [[] for _ in range(n)]
+ chunk_indices = [[] for _ in range(n)]
+ for i, req in enumerate(requests):
+ slot = i % n
+ chunks[slot].append(req)
+ chunk_indices[slot].append(i)
+
+ results = [None] * len(requests)
+ for i in range(n):
+ if not chunks[i]:
+ continue
+ responses = self._batch_with_retry(self._models[i], chunks[i])
+ for idx, response in zip(chunk_indices[i], responses):
+ results[idx] = response
+ return results
+
+ def _batch_with_retry(self, model, requests: list, max_attempts: int = 4) -> list:
+ c = self._MAX_INNER_CONCURRENCY
+ for attempt in range(max_attempts):
+ try:
+ return model.batch(requests, config={"max_concurrency": c})
+ except Exception:
+ if attempt < max_attempts - 1:
+ c = max(1, c // 2)
+ time.sleep(5 * (attempt + 1))
+ else:
+ raise
+
+ def invoke(self, request):
+ return self._models[0].invoke(request)
+
+ def __getattr__(self, name):
+ return getattr(self._models[0], name)
+
+
+def load_proxy_list(proxy_config: dict) -> list:
+ proxy_list = proxy_config.get("proxies") or []
+ if not proxy_list:
+ legacy = proxy_config.get("proxy", {}).get("http")
+ if legacy:
+ proxy_list = [legacy]
+ return proxy_list
+
+
+def normalize_model_name(provider: str, model_name: str) -> str:
+ if provider == "openrouter" and "/" not in model_name:
+ return f"openai/{model_name}"
+ if provider == "openai" and model_name.startswith("openai/"):
+ return model_name.split("/", 1)[1]
+ return model_name
+
+
+def _resolve_api_keys(config: dict) -> list:
+ all_keys = config.get("openai_api_keys") or (
+ [config["openai_api_key"]] if config.get("openai_api_key") else [os.getenv("OPENAI_API_KEY")]
+ )
+ active = config.get("active_keys")
+ keys = [all_keys[i] for i in active if i < len(all_keys)] if active is not None else all_keys
+ return [k for k in keys if k]
+
+
+def _build_models(api_keys, model_name, base_url, temperature, proxy_list, model_kwargs, requests_per_minute=None):
+ models = []
+ for idx, key in enumerate(api_keys):
+ key_http_client = None
+ if proxy_list:
+ proxy_url = proxy_list[idx % len(proxy_list)]
+ key_http_client = httpx.Client(proxy=proxy_url, timeout=60.0)
+ rate_limiter = None
+ if requests_per_minute:
+ rate_limiter = InMemoryRateLimiter(
+ requests_per_second=requests_per_minute / 60.0,
+ check_every_n_seconds=0.1,
+ max_bucket_size=requests_per_minute,
+ )
+ models.append(ChatOpenAI(
+ model=model_name,
+ api_key=key,
+ base_url=base_url,
+ temperature=temperature,
+ rate_limiter=rate_limiter,
+ max_retries=3,
+ model_kwargs=model_kwargs,
+ http_client=key_http_client,
+ ))
+ return MultiKeyModel(models) if len(models) > 1 else models[0]
+
+
+def create_model(provider, model_name, requests_per_minute, config, proxy_config, temperature=0.0):
+ proxy_list = load_proxy_list(proxy_config)
+ model_name = normalize_model_name(provider, model_name)
+
+ if provider == "openrouter":
+ api_keys = [config.get("openrouter_api_key") or os.getenv("OPENROUTER_API_KEY")]
+ base_url = "https://openrouter.ai/api/v1"
+ model_kwargs = {"extra_body": {"provider": {"order": ["openai"], "allow_fallbacks": False}}}
+ else:
+ api_keys = _resolve_api_keys(config)
+ base_url = None
+ model_kwargs = {}
+
+ api_keys = [k for k in api_keys if k]
+ if not api_keys:
+ raise ValueError("No API keys found in config")
+
+ if requests_per_minute:
+ print(f"{len(api_keys)} keys, {requests_per_minute} RPM (total {len(api_keys) * requests_per_minute})")
+ if proxy_list:
+ print(f"{len(proxy_list)} proxies")
+
+ return _build_models(api_keys, model_name, base_url, temperature, proxy_list, model_kwargs, requests_per_minute)
diff --git a/notebooks/experiments/pipeline/optmize_single.py b/notebooks/experiments/pipeline/optmize_single.py
new file mode 100644
index 00000000..7d148e1b
--- /dev/null
+++ b/notebooks/experiments/pipeline/optmize_single.py
@@ -0,0 +1,452 @@
+import os
+import sys
+import json
+import yaml
+import time
+import random
+import argparse
+import traceback
+from datetime import datetime
+
+_script_dir = os.path.dirname(os.path.abspath(__file__))
+sys.path.append(os.path.abspath(os.path.join(_script_dir, "../../../")))
+sys.path.append(os.path.abspath(os.path.join(_script_dir, "../../../src")))
+
+import numpy as np
+import torch
+import gc
+
+from coolprompt.optimizer.reflective_prompt.evoluter import ReflectiveEvoluter
+from coolprompt.optimizer.reflective_prompt.factorized_evoluter import (
+ FactorizedEvoluter,
+)
+from coolprompt.optimizer.reflective_prompt.coevo_evoluter import (
+ CoevoEvoluter,
+ PerFieldCoevoEvoluter,
+)
+from coolprompt.evaluator import Evaluator, validate_and_create_metric
+from coolprompt.utils.logging_config import setup_logging
+
+from model_utils import create_model
+from dataset_config import DATASETS_CONFIG, load_train_data
+
+setup_logging()
+
+_EVAL_CONFIG_PATH = os.path.join(_script_dir, "evaluation_config.yaml")
+_CONFIG_PATH = os.path.join(_script_dir, "config.yaml")
+_PROXY_CONFIG_PATH = os.path.join(_script_dir, "proxy_config.yaml")
+
+with open(_CONFIG_PATH) as f:
+ CONFIG = yaml.safe_load(f)
+
+PROXY_CONFIG = (
+ yaml.safe_load(open(_PROXY_CONFIG_PATH))
+ if os.path.exists(_PROXY_CONFIG_PATH)
+ else {}
+)
+
+if CONFIG.get("openai_api_key"):
+ os.environ["OPENAI_API_KEY"] = CONFIG["openai_api_key"]
+if CONFIG.get("openrouter_api_key"):
+ os.environ["OPENROUTER_API_KEY"] = CONFIG["openrouter_api_key"]
+
+_FACTORIZED_MODES = {"factorized", "factorized_dedup", "factorized_top_prompts"}
+_VALID_ROLE_MODES = {
+ "with_role",
+ "no_role",
+ "coevo",
+ "coevo_enhanced",
+ "coevo_no_enhancements",
+ "coevo_per_field",
+} | _FACTORIZED_MODES
+
+
+def _update_eval_config(dataset_name: str, result_file: str) -> None:
+ with open(_EVAL_CONFIG_PATH) as f:
+ eval_cfg = yaml.safe_load(f)
+ rel_path = os.path.relpath(result_file, os.path.dirname(_EVAL_CONFIG_PATH))
+ if os.sep != "/":
+ rel_path = rel_path.replace(os.sep, "/")
+ if "dataset_paths" not in eval_cfg or eval_cfg["dataset_paths"] is None:
+ eval_cfg["dataset_paths"] = {}
+ eval_cfg["dataset_paths"][dataset_name] = [rel_path]
+ with open(_EVAL_CONFIG_PATH, "w") as f:
+ yaml.dump(eval_cfg, f, allow_unicode=True, sort_keys=False)
+ print(f"evaluation_config.yaml updated: {dataset_name} -> {rel_path}")
+
+
+def run_optimization(
+ args,
+ config,
+ train_inputs,
+ train_targets,
+ val_inputs,
+ val_targets,
+ logs_dir,
+ role_mode,
+ settings,
+):
+ temperature = settings["temperature"]
+ model = create_model(
+ args.provider,
+ args.model,
+ args.requests_per_minute,
+ CONFIG,
+ PROXY_CONFIG,
+ temperature=temperature,
+ )
+ val_model = create_model(
+ args.provider, args.model, None, CONFIG, PROXY_CONFIG, temperature=0.0
+ )
+
+ metric = validate_and_create_metric(config["task"], config["metric"])
+ evaluator = Evaluator(model, config["task"], metric)
+ val_evaluator = Evaluator(val_model, config["task"], metric)
+
+ pop_size = settings["population_size"]
+ num_epochs = settings["num_epochs"]
+ phase_epochs = settings["factorized_phase_epochs"]
+ use_enhancements = settings["use_enhancements"]
+ use_dedup = settings["use_dedup"]
+ evolve_constraints = settings["evolve_constraints"]
+ task_desc = config["initial_task_description"]
+ initial_constraints = config.get("initial_output_constraints", "")
+
+ print(f"\nrunning: {role_mode}")
+
+ if role_mode in _FACTORIZED_MODES:
+ evoluter = FactorizedEvoluter(
+ model=model,
+ evaluator=evaluator,
+ train_dataset=train_inputs,
+ train_targets=train_targets,
+ validation_dataset=val_inputs,
+ validation_targets=val_targets,
+ problem_description=f"Task: {config['description']}",
+ initial_prompt=task_desc,
+ initial_role=config["initial_system_behavior"],
+ initial_constraints=(
+ initial_constraints if evolve_constraints else None
+ ),
+ population_size=pop_size,
+ phase_epochs=phase_epochs,
+ run_constraints_phase=evolve_constraints,
+ use_cache=True,
+ output_path=logs_dir,
+ use_enhancements=use_enhancements,
+ use_dedup=use_dedup,
+ val_evaluator=val_evaluator,
+ )
+ elif role_mode in ("coevo_enhanced", "coevo_no_enhancements"):
+ evoluter = CoevoEvoluter(
+ model=model,
+ evaluator=evaluator,
+ train_dataset=train_inputs,
+ train_targets=train_targets,
+ validation_dataset=val_inputs,
+ validation_targets=val_targets,
+ problem_description=f"Task: {config['description']}",
+ initial_prompt=task_desc,
+ initial_role=config["initial_system_behavior"],
+ initial_constraints=initial_constraints,
+ population_size=pop_size,
+ num_epochs=num_epochs,
+ use_cache=True,
+ output_path=logs_dir,
+ use_enhancements=(role_mode == "coevo_enhanced"),
+ val_evaluator=val_evaluator,
+ )
+ elif role_mode == "coevo_per_field":
+ evoluter = PerFieldCoevoEvoluter(
+ model=model,
+ evaluator=evaluator,
+ train_dataset=train_inputs,
+ train_targets=train_targets,
+ validation_dataset=val_inputs,
+ validation_targets=val_targets,
+ problem_description=f"Task: {config['description']}",
+ initial_prompt=task_desc,
+ initial_role=config["initial_system_behavior"],
+ initial_constraints=initial_constraints,
+ population_size=pop_size,
+ num_epochs=num_epochs,
+ use_cache=True,
+ output_path=logs_dir,
+ use_enhancements=True,
+ val_evaluator=val_evaluator,
+ )
+ else:
+ if role_mode == "coevo":
+ initial_role, evolve_role = config["initial_system_behavior"], True
+ elif role_mode == "with_role":
+ initial_role, evolve_role = config["initial_system_behavior"], False
+ else:
+ initial_role, evolve_role = "", False
+ evoluter = ReflectiveEvoluter(
+ model=model,
+ evaluator=evaluator,
+ train_dataset=train_inputs,
+ train_targets=train_targets,
+ validation_dataset=val_inputs,
+ validation_targets=val_targets,
+ problem_description=f"Task: {config['description']}",
+ initial_prompt=task_desc,
+ initial_role=initial_role,
+ initial_constraints=(
+ initial_constraints if evolve_constraints else None
+ ),
+ evolve_role=evolve_role,
+ evolve_constraints=evolve_constraints,
+ population_size=pop_size,
+ num_epochs=num_epochs,
+ use_cache=True,
+ output_path=logs_dir,
+ use_enhancements=use_enhancements,
+ val_evaluator=val_evaluator,
+ )
+
+ evoluter.evolution()
+ return evoluter
+
+
+def main():
+ parser = argparse.ArgumentParser(
+ description="Optimize prompt for multiple datasets"
+ )
+ parser.add_argument(
+ "--provider",
+ type=str,
+ default=CONFIG["provider"],
+ choices=["openai", "openrouter"],
+ )
+ parser.add_argument("--model", type=str, default=CONFIG["model"])
+ _output_dir = CONFIG["output_dir"]
+ if not os.path.isabs(_output_dir):
+ _output_dir = os.path.abspath(os.path.join(_script_dir, _output_dir))
+ parser.add_argument("--output_dir", type=str, default=_output_dir)
+ parser.add_argument("--requests_per_minute", type=int, default=None)
+ parser.add_argument(
+ "--debug",
+ action="store_true",
+ help="Minimal sizes (train=5, val=5, pop=2, epochs=1)",
+ )
+ args = parser.parse_args()
+
+ if args.requests_per_minute is None:
+ args.requests_per_minute = CONFIG["requests_per_minute"].get(
+ args.provider, 500
+ )
+
+ settings = {
+ "population_size": CONFIG["population_size"],
+ "num_epochs": CONFIG["num_epochs"],
+ "train_size": CONFIG["train_size"],
+ "val_size": CONFIG["val_size"],
+ "factorized_phase_epochs": tuple(
+ CONFIG.get("factorized_phase_epochs", [4, 3, 3])
+ ),
+ "temperature": CONFIG.get("temperature", 0.0),
+ "use_enhancements": CONFIG.get("use_enhancements", True),
+ "use_dedup": CONFIG.get("use_dedup", True),
+ "evolve_constraints": CONFIG.get("evolve_constraints", False),
+ }
+
+ if args.debug:
+ settings.update(
+ {
+ "population_size": 2,
+ "num_epochs": 1,
+ "train_size": 5,
+ "val_size": 5,
+ "factorized_phase_epochs": (1, 1, 1),
+ }
+ )
+ print("Debug mode: train=5, val=5, pop=2, epochs=1")
+
+ _seed = CONFIG.get("seed", 42)
+ random.seed(_seed)
+ np.random.seed(_seed)
+
+ datasets_to_run = CONFIG.get("datasets_to_run", [])
+ role_modes_to_run = CONFIG.get("role_modes_to_run", ["with_role"])
+
+ print(f"Datasets: {datasets_to_run}")
+ print(f"Role modes: {role_modes_to_run}")
+ print(
+ f"Provider: {args.provider}, Model: {args.model}, RPM: {args.requests_per_minute}"
+ )
+ print(f"Enhancements: {'ON' if settings['use_enhancements'] else 'OFF'}")
+ print(
+ f"Constraints evolution: {'ON' if settings['evolve_constraints'] else 'OFF'}"
+ )
+
+ for dataset_name in datasets_to_run:
+ if dataset_name not in DATASETS_CONFIG:
+ print(f"skip unknown dataset: {dataset_name}")
+ continue
+
+ print(f"\n{dataset_name}")
+
+ config = DATASETS_CONFIG[dataset_name]
+ run_dir = os.path.join(args.output_dir, dataset_name)
+ os.makedirs(run_dir, exist_ok=True)
+
+ train_size = settings["train_size"]
+ val_size = settings["val_size"]
+ try:
+ inputs, targets = load_train_data(
+ dataset_name, config, num_samples=train_size + val_size + 20
+ )
+ except Exception as e:
+ print(f"failed to load {dataset_name}: {e}")
+ continue
+
+ if len(inputs) < train_size + val_size:
+ print(
+ f"warning: only {len(inputs)} samples, need {train_size + val_size}"
+ )
+
+ train_inputs = inputs[:train_size]
+ train_targets = targets[:train_size]
+ val_inputs = inputs[train_size : train_size + val_size]
+ val_targets = targets[train_size : train_size + val_size]
+ print(f"Split: train={len(train_inputs)}, val={len(val_inputs)}")
+
+ for role_mode in role_modes_to_run:
+ if role_mode not in _VALID_ROLE_MODES:
+ print(f"skip unknown mode: {role_mode}")
+ continue
+
+ method_dir = os.path.join(run_dir, role_mode)
+ os.makedirs(method_dir, exist_ok=True)
+
+ timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
+ logs_dir = os.path.join(
+ method_dir, "logs", f"logs_{role_mode}_{timestamp}"
+ )
+ os.makedirs(logs_dir, exist_ok=True)
+ print(f"\nmode: {role_mode}")
+ print(f"Logs: {logs_dir}")
+
+ max_retries = 3
+ evoluter = None
+ start_time = time.time()
+
+ for attempt in range(max_retries):
+ try:
+ if attempt > 0:
+ print(f"Retry {attempt + 1}/{max_retries}...")
+ time.sleep(60)
+ evoluter = run_optimization(
+ args,
+ config,
+ train_inputs,
+ train_targets,
+ val_inputs,
+ val_targets,
+ logs_dir,
+ role_mode,
+ settings,
+ )
+ break
+ except Exception as e:
+ error_msg = str(e)
+ print(
+ f"error {dataset_name}/{role_mode} (attempt {attempt + 1}): {error_msg}"
+ )
+ if (
+ "429" in error_msg
+ or "Rate limit" in error_msg
+ or "quota" in error_msg
+ ):
+ wait_time = (attempt + 1) * 60
+ print(f"Rate limit. Waiting {wait_time}s...")
+ time.sleep(wait_time)
+ else:
+ time.sleep(30)
+ if attempt == max_retries - 1:
+ print(f"max retries: {dataset_name}/{role_mode}")
+ with open(
+ os.path.join(
+ method_dir,
+ f"error_log_{role_mode}_{timestamp}.txt",
+ ),
+ "w",
+ ) as f:
+ f.write(
+ f"Failed after {max_retries} attempts.\nLast error: {error_msg}\n{traceback.format_exc()}"
+ )
+
+ duration = time.time() - start_time
+
+ if evoluter:
+ is_factorized = role_mode in _FACTORIZED_MODES
+ result_data = {
+ "dataset": dataset_name,
+ "role_mode": role_mode,
+ "model": args.model,
+ "best_prompt": evoluter.best_prompt_overall,
+ "best_role": evoluter.best_role_overall,
+ "best_constraints": evoluter.best_constraints_overall or "",
+ "best_score": evoluter.best_score_overall,
+ "candidates": getattr(evoluter, "candidates", None),
+ "initial_task_description": evoluter.initial_prompt,
+ "initial_system_behavior": evoluter.initial_role or "",
+ "initial_output_constraints": evoluter.initial_constraints
+ or "",
+ "description": config["description"],
+ "parameters": {
+ "population_size": settings["population_size"],
+ "num_epochs": (
+ None if is_factorized else settings["num_epochs"]
+ ),
+ "factorized_phase_epochs": (
+ list(settings["factorized_phase_epochs"])
+ if is_factorized
+ else None
+ ),
+ "train_size": len(train_inputs),
+ "val_size": len(val_inputs),
+ "rate_limit_rpm": args.requests_per_minute,
+ "provider": args.provider,
+ "temperature": settings["temperature"],
+ "val_temperature": 0.0,
+ "use_enhancements": (
+ (role_mode == "coevo_enhanced")
+ if role_mode
+ in ("coevo_enhanced", "coevo_no_enhancements")
+ else settings["use_enhancements"]
+ ),
+ "evolve_constraints": settings["evolve_constraints"],
+ },
+ "duration_seconds": duration,
+ "timestamp": timestamp,
+ }
+
+ score = evoluter.best_score_overall
+ score_str = (
+ f"{score:.2f}" if isinstance(score, (int, float)) else "NA"
+ )
+ result_filename = (
+ f"{timestamp}_{score_str}_{role_mode}_seed{_seed}.json"
+ )
+ result_file = os.path.join(method_dir, result_filename)
+ with open(result_file, "w") as f:
+ json.dump(result_data, f, indent=2)
+
+ _update_eval_config(dataset_name, result_file)
+
+ print(
+ f"\ndone {dataset_name}/{role_mode}, score={evoluter.best_score_overall}"
+ )
+ print(f"saved: {result_file}")
+
+ del evoluter
+ gc.collect()
+ torch.cuda.empty_cache()
+
+ print("\ndone")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/requirements.txt b/requirements.txt
index d18ee6fb..2c9251d7 100644
--- a/requirements.txt
+++ b/requirements.txt
@@ -15,4 +15,5 @@ langchain_huggingface>=0.3.1
langchain-openai>=0.3.30
langdetect>=0.4.31
deepeval>=3.7.2
-transformers<5.0.0
\ No newline at end of file
+transformers<5.0.0
+sentence-transformers>=3.0.0
\ No newline at end of file