diff --git a/README.md b/README.md index 050c64d9..60534512 100644 --- a/README.md +++ b/README.md @@ -15,6 +15,108 @@ [![ITMO](https://raw.githubusercontent.com/aimclub/open-source-ops/43bb283758b43d75ec1df0a6bb4ae3eb20066323/badges/ITMO_badge.svg)](https://itmo.ru/) [![Telegram Channel](https://img.shields.io/badge/Telegram-2CA5E0?style=flat&logo=telegram&logoColor=white)](https://t.me/+0kMcymeAQrczN2Fi) +--- + +
+ +# 🧬 CoEvo β€” структурная ΠΊΠΎ-ΡΠ²ΠΎΠ»ΡŽΡ†ΠΈΡ ΠΏΡ€ΠΎΠΌΠΏΡ‚ΠΎΠ² + +**Выпускная квалификационная Ρ€Π°Π±ΠΎΡ‚Π°** + +ΠœΠ΅Ρ‚ΠΎΠ΄ автоматичСской ΠΎΠΏΡ‚ΠΈΠΌΠΈΠ·Π°Ρ†ΠΈΠΈ ΠΏΡ€ΠΎΠΌΠΏΡ‚ΠΎΠ², Ρ€Π΅Π°Π»ΠΈΠ·ΠΎΠ²Π°Π½Π½Ρ‹ΠΉ Π²Π½ΡƒΡ‚Ρ€ΠΈ Ρ„Ρ€Π΅ΠΉΠΌΠ²ΠΎΡ€ΠΊΠ° CoolPrompt + +
+ +> Π­Ρ‚ΠΎΡ‚ Ρ€Π°Π·Π΄Π΅Π» (Π²Π΅Ρ‚ΠΊΠ° `role_based`) описываСт ΠΌΠΎΠΉ Π΄ΠΈΠΏΠ»ΠΎΠΌΠ½Ρ‹ΠΉ ΠΌΠ΅Ρ‚ΠΎΠ΄ **CoEvo** ΠΈ Π΅Π³ΠΎ ΡƒΡΠΈΠ»Π΅Π½Π½ΡƒΡŽ Π²Π΅Ρ€ΡΠΈΡŽ **CoEvo-M**: идСю, Ρ€Π΅Π·ΡƒΠ»ΡŒΡ‚Π°Ρ‚Ρ‹, запускаСмоС Π΄Π΅ΠΌΠΎ ΠΈ список ΠΊΠ»ΡŽΡ‡Π΅Π²Ρ‹Ρ… Ρ„Π°ΠΉΠ»ΠΎΠ². ΠžΠ±Ρ‰Π΅Π΅ описаниС Ρ„Ρ€Π΅ΠΉΠΌΠ²ΠΎΡ€ΠΊΠ° CoolPrompt β€” [Π½ΠΈΠΆΠ΅](#coolprompt-framework). + +## Π’ Ρ‡Ρ‘ΠΌ идСя + +ΠšΠ»Π°ΡΡΠΈΡ‡Π΅ΡΠΊΠΈΠ΅ ΠΌΠ΅Ρ‚ΠΎΠ΄Ρ‹ ΠΎΠΏΡ‚ΠΈΠΌΠΈΠ·ΠΈΡ€ΡƒΡŽΡ‚ ΠΏΡ€ΠΎΠΌΠΏΡ‚ ΠΊΠ°ΠΊ **Π΅Π΄ΠΈΠ½Ρ‹ΠΉ кусок тСкста**. CoEvo прСдставляСт ΠΏΡ€ΠΎΠΌΠΏΡ‚ ΠΊΠ°ΠΊ **Ρ‚Ρ€ΠΈ нСзависимых поля** ΠΈ ΡΠ²ΠΎΠ»ΡŽΡ†ΠΈΠΎΠ½ΠΈΡ€ΡƒΠ΅Ρ‚ ΠΊΠ°ΠΆΠ΄ΠΎΠ΅ ΠΈΠ· Π½ΠΈΡ…: + +| ПолС | ΠšΡƒΠ΄Π° подставляСтся | Π—Π° Ρ‡Ρ‚ΠΎ ΠΎΡ‚Π²Π΅Ρ‡Π°Π΅Ρ‚ | +|------|--------------------|-----------------| +| `role` (`system_behavior`) | **system**-сообщСниС | Ρ€ΠΎΠ»ΡŒ ΠΈ ΠΏΠΎΠ²Π΅Π΄Π΅Π½ΠΈΠ΅ ΠΌΠΎΠ΄Π΅Π»ΠΈ | +| `task` (`task_description`) | **user**-сообщСниС | Ρ‡Ρ‚ΠΎ ΠΈΠΌΠ΅Π½Π½ΠΎ Π½ΡƒΠΆΠ½ΠΎ ΡΠ΄Π΅Π»Π°Ρ‚ΡŒ | +| `constraints` (`output_constraints`) | **user**-сообщСниС | Ρ„ΠΎΡ€ΠΌΠ°Ρ‚ ΠΈ ограничСния ΠΎΡ‚Π²Π΅Ρ‚Π° | + +1. **ДСкомпозиция.** На стартС ΠΎΠ΄ΠΈΠ½ Π²Ρ‹Π·ΠΎΠ² LLM-ΠΎΠΏΡ‚ΠΈΠΌΠΈΠ·Π°Ρ‚ΠΎΡ€Π° раскладываСт исходный ΠΏΡ€ΠΎΠΌΠΏΡ‚ Π½Π° Ρ‚Ρ€ΠΎΠΉΠΊΡƒ `role / task / constraints` (структурированный JSON). +2. **Π­Π²ΠΎΠ»ΡŽΡ†ΠΈΡ.** ΠŸΠΎΠΏΡƒΠ»ΡΡ†ΠΈΡ Ρ‚Π°ΠΊΠΈΡ… Ρ‚Ρ€ΠΎΠ΅ΠΊ оптимизируСтся гСнСтичСски: Ρ€ΡƒΠ»Π΅Ρ‚ΠΎΡ‡Π½Ρ‹ΠΉ ΠΎΡ‚Π±ΠΎΡ€ β†’ рСфлСксия β†’ кроссовСр β†’ мутация β†’ softmax-Π²Ρ‹ΠΆΠΈΠ²Π°Π½ΠΈΠ΅. РСфлСксия ΠΎΠ±ΡŠΡΡΠ½ΡΠ΅Ρ‚ ΠΌΠΎΠ΄Π΅Π»ΠΈ, *Ρ‡Π΅ΠΌ* ΡƒΠ΄Π°Ρ‡Π½Ρ‹Π΅ Π²Π°Ρ€ΠΈΠ°Π½Ρ‚Ρ‹ Π»ΡƒΡ‡ΡˆΠ΅ Π½Π΅ΡƒΠ΄Π°Ρ‡Π½Ρ‹Ρ…. +3. **ΠžΡ‚Π±ΠΎΡ€ ΠΏΠΎΠ»Π΅ΠΉ.** Π’ ΠΊΠΎΠ½Ρ†Π΅ ablation Π½Π° Π²Π°Π»ΠΈΠ΄Π°Ρ†ΠΈΠΈ Π²Ρ‹Π±ΠΈΡ€Π°Π΅Ρ‚ Π»ΡƒΡ‡ΡˆΡƒΡŽ ΠΊΠΎΠΌΠ±ΠΈΠ½Π°Ρ†ΠΈΡŽ ΠΏΠΎΠ»Π΅ΠΉ (task / task+role / task+role+constraints). + +**CoEvo-M** β€” усилСнная вСрсия: ΡˆΡ‚Ρ€Π°Ρ„ Π·Π° Π΄Π»ΠΈΠ½Ρƒ Ρ€ΠΎΠ»ΠΈ ΠΈ Π·Π° смысловоС Π΄ΡƒΠ±Π»ΠΈΡ€ΠΎΠ²Π°Π½ΠΈΠ΅ `role`/`task` (sentence-transformers), hall-of-fame Π»ΡƒΡ‡ΡˆΠΈΡ… особСй, Β«ΠΏΠ»ΠΎΡ…ΠΈΠ΅ ΠΏΡ€ΠΈΠΌΠ΅Ρ€Ρ‹Β» Π² ΠΌΡƒΡ‚Π°Ρ†ΠΈΠΈ ΠΈ форсированный элитизм. + +## Π Π΅Π·ΡƒΠ»ΡŒΡ‚Π°Ρ‚Ρ‹ + +Π‘Ρ€Π°Π²Π½Π΅Π½ΠΈΠ΅ с Π±Π°Π·ΠΎΠ²Ρ‹ΠΌ ReflectivePrompt Π½Π° 6 датасСтах (BERTScore / ΠΌΠ΅Ρ‚Ρ€ΠΈΠΊΠ° Π·Π°Π΄Π°Ρ‡ΠΈ): + +

+ CoEvo benchmark +

+ +| ДатасСт | ReflectivePrompt | CoEvo | CoEvo-M | +|---------|:---:|:---:|:---:| +| TweetEval | 0.705 | **0.726** | 0.719 | +| SQuAD v2 | 0.878 | 0.907 | **0.929** | +| CommonGen | 0.808 | **0.809** | 0.807 | +| MEDIQA | 0.688 | 0.700 | **0.703** | +| GSM8K | 0.919 | **0.927** | 0.926 | +| XSum | 0.730 | **0.736** | 0.734 | +| **Π‘Ρ€Π΅Π΄Π½Π΅Π΅** | 0.788 | 0.801 | **0.803** | + +## ЗапускаСмоС Π΄Π΅ΠΌΠΎ + +πŸ““ **[notebooks/examples/coevo_demo.ipynb](notebooks/examples/coevo_demo.ipynb)** β€” CoEvo end-to-end ΡƒΠ»ΡƒΡ‡ΡˆΠ°Π΅Ρ‚ ΠΏΡ€ΠΎΠΌΠΏΡ‚ для QA ΠΏΠΎ SQuAD v2. + +ИдСя сцСнария: ΡΠΈΠ»ΡŒΠ½Ρ‹ΠΉ ΠΎΠΏΡ‚ΠΈΠΌΠΈΠ·Π°Ρ‚ΠΎΡ€ (`gpt-4o-mini`) пСрСписываСт ΠΏΡ€ΠΎΠΌΠΏΡ‚ для Π΄Π΅ΡˆΡ‘Π²ΠΎΠΉ ΠΏΡ€ΠΎΠ΄Π°ΠΊΡˆΠ½-ΠΌΠΎΠ΄Π΅Π»ΠΈ (`gpt-4.1-nano`). + +

+ CoEvo demo result +

+ +Из простого `"Answer the question based on the context."` ΠΌΠ΅Ρ‚ΠΎΠ΄ Π·Π° 5 эпох собираСт структурированный ΠΏΡ€ΠΎΠΌΠΏΡ‚ ΠΈ ΠΏΠΎΠ΄Π½ΠΈΠΌΠ°Π΅Ρ‚ BERTScore **0.823 β†’ 0.896 (+0.073)**. + +## Быстрый старт CoEvo + +```python +from coolprompt.assistant import PromptTuner + +tuner = PromptTuner() # OPENAI_API_KEY Π² ΠΎΠΊΡ€ΡƒΠΆΠ΅Π½ΠΈΠΈ + +tuner.run( + start_prompt="Answer the question based on the context.", + task="generation", + metric="bertscore", + dataset=dataset, # список Π²Ρ…ΠΎΠ΄ΠΎΠ² + target=target, # список эталонных ΠΎΡ‚Π²Π΅Ρ‚ΠΎΠ² + method="coevo", # CoEvo-M ΠΏΠΎ ΡƒΠΌΠΎΠ»Ρ‡Π°Π½ΠΈΡŽ; use_enhancements=False Π±Π°Π·ΠΎΠ²Ρ‹ΠΉ CoEvo +) + +# CoEvo Π²ΠΎΠ·Π²Ρ€Π°Ρ‰Π°Π΅Ρ‚ Ρ‚Ρ€ΠΈ поля: +print(tuner.final_role) # Ρ€ΠΎΠ»ΡŒ +print(tuner.final_prompt) # Π·Π°Π΄Π°Ρ‡Π° +print(tuner.final_constraints) # ограничСния Ρ„ΠΎΡ€ΠΌΠ°Ρ‚Π° +``` + +## Мой Π²ΠΊΠ»Π°Π΄ ΠΈ ΠΊΠ»ΡŽΡ‡Π΅Π²Ρ‹Π΅ Ρ„Π°ΠΉΠ»Ρ‹ + +РСализация ΠΌΠ΅Ρ‚ΠΎΠ΄ΠΎΠ² CoEvo / CoEvo-M ΠΏΠΎΠ²Π΅Ρ€Ρ… Ρ„Ρ€Π΅ΠΉΠΌΠ²ΠΎΡ€ΠΊΠ° CoolPrompt: + +- **[`coolprompt/optimizer/reflective_prompt/coevo_base_evoluter.py`](coolprompt/optimizer/reflective_prompt/coevo_base_evoluter.py)** β€” ядро ΠΌΠ΅Ρ‚ΠΎΠ΄Π°: дСкомпозиция ΠΏΡ€ΠΎΠΌΠΏΡ‚Π° Π½Π° 3 поля, ΡΠ²ΠΎΠ»ΡŽΡ†ΠΈΠΎΠ½Π½Ρ‹ΠΉ Ρ†ΠΈΠΊΠ», рСфлСксия, ΠΎΡ‚Π±ΠΎΡ€ ΠΏΠΎΠ»Π΅ΠΉ. +- **[`coolprompt/optimizer/reflective_prompt/coevo_evoluter.py`](coolprompt/optimizer/reflective_prompt/coevo_evoluter.py)** β€” ΠΎΠΏΠ΅Ρ€Π°Ρ‚ΠΎΡ€Ρ‹ кроссовСра ΠΈ ΠΌΡƒΡ‚Π°Ρ†ΠΈΠΈ Π½Π° ΡƒΡ€ΠΎΠ²Π½Π΅ ΠΏΠΎΠ»Π΅ΠΉ. +- **[`coolprompt/optimizer/reflective_prompt/factorized_evoluter.py`](coolprompt/optimizer/reflective_prompt/factorized_evoluter.py)** β€” факторизованная ΡΠ²ΠΎΠ»ΡŽΡ†ΠΈΡ ΠΏΠΎ ΠΎΡ‚Π΄Π΅Π»ΡŒΠ½Ρ‹ΠΌ полям. +- **[`coolprompt/optimizer/reflective_prompt/run.py`](coolprompt/optimizer/reflective_prompt/run.py)** β€” `CoevoMethod`, интСграция Π² ΠΏΡƒΠ±Π»ΠΈΡ‡Π½Ρ‹ΠΉ API (`method="coevo"`). +- **[`coolprompt/utils/prompt_templates/`](coolprompt/utils/prompt_templates/)** β€” ΠΌΠ΅Ρ‚Π°-ΠΏΡ€ΠΎΠΌΠΏΡ‚Ρ‹ CoEvo: `reflective_templates_coevo_enhanced.py`, `reflective_templates_coevo_per_field.py`, `reflective_templates_coevolution.py`. + + +## ΠœΠ°Ρ‚Π΅Ρ€ΠΈΠ°Π»Ρ‹ +- πŸ““ Π”Π΅ΠΌΠΎ-Π½ΠΎΡƒΡ‚Π±ΡƒΠΊ: [coevo_demo.ipynb](notebooks/examples/coevo_demo.ipynb). + +--- + + + +# CoolPrompt β€” Ρ„Ρ€Π΅ΠΉΠΌΠ²ΠΎΡ€ΠΊ Π°Π²Ρ‚ΠΎΠΏΡ€ΠΎΠΌΠΏΡ‚ΠΈΠ½Π³Π° + CoolPrompt is a framework for automatic prompt creation and optimization. ### Join our [telegram](https://t.me/+0kMcymeAQrczN2Fi) channel to be in touch. diff --git a/coolprompt/assistant.py b/coolprompt/assistant.py index 6c86b101..9345fde0 100644 --- a/coolprompt/assistant.py +++ b/coolprompt/assistant.py @@ -61,6 +61,8 @@ def __init__( self.init_prompt = None self.final_metric = None self.final_prompt = None + self.final_role = None + self.final_constraints = None self.assistant_feedback = None self.synthetic_dataset = None @@ -323,6 +325,11 @@ def run( **kwargs, ) + self.final_role = getattr(method_impl, "last_role", "") or None + self.final_constraints = ( + getattr(method_impl, "last_constraints", "") or None + ) + logger.info("Running the prompt format checking...") final_prompt = correct( prompt=final_prompt, @@ -346,6 +353,8 @@ def run( dataset=dataset_split[1], targets=dataset_split[3], template=template, + system_role=self.final_role, + constraints=self.final_constraints, ) logger.info( f"Initial {base_metric} score: {self.init_metric}, " diff --git a/coolprompt/evaluator/evaluator.py b/coolprompt/evaluator/evaluator.py index f6f059dc..283df7f5 100644 --- a/coolprompt/evaluator/evaluator.py +++ b/coolprompt/evaluator/evaluator.py @@ -6,6 +6,7 @@ from langchain_core.language_models.base import BaseLanguageModel from langchain_core.messages.ai import AIMessage +from langchain_core.messages import SystemMessage, HumanMessage import numpy as np from coolprompt.evaluator.metrics import BaseMetric from coolprompt.utils.logging_config import logger @@ -67,6 +68,8 @@ def evaluate( targets: list[str | int], template: Optional[str] = None, failed_examples: Optional[int] = None, + system_role: Optional[str] = None, + constraints: Optional[str] = None, *, return_detailed: bool = False, save_model_answers: bool = False, @@ -112,7 +115,9 @@ def evaluate( if self.task == Task.CLASSIFICATION: self.metric.extract_labels(targets) full_prompts = [ - self._get_full_prompt(prompt, sample, template) + self._get_full_prompt( + prompt, sample, template, system_role, constraints + ) for sample in dataset ] @@ -203,7 +208,9 @@ def _get_full_prompt( prompt: str, sample: str, template: Optional[str] = None, - ) -> str: + system_role: Optional[str] = None, + constraints: Optional[str] = None, + ) -> str | list: """Inserts parts of the prompt into the task template. Args: @@ -212,25 +219,43 @@ def _get_full_prompt( template (Optional[str]): Prompt template for defined task type. If None, uses default template. + system_role (Optional[str]): system behavior prepended as a + SystemMessage (CoEvo). Defaults to None. + constraints (Optional[str]): output format constraints appended + to the prompt (CoEvo). Defaults to None. Raises: ValueError: if type of task is not supported Returns: - str: the full prompt to be passed to the model + str | list: the full prompt string, or a list of + SystemMessage + HumanMessage if system_role is set. """ if template is None: template = self._get_default_template() + effective_prompt = prompt + if constraints: + effective_prompt = f"{prompt}\n\n{constraints}" + match self.task: case Task.CLASSIFICATION: labels = ", ".join(map(str, self.metric.label_to_id.keys())) - return template.format( - PROMPT=prompt, LABELS=labels, INPUT=sample + formatted = template.format( + PROMPT=effective_prompt, LABELS=labels, INPUT=sample ) case Task.GENERATION: - return template.format(PROMPT=prompt, INPUT=sample) + formatted = template.format( + PROMPT=effective_prompt, INPUT=sample + ) + + if system_role: + return [ + SystemMessage(content=system_role), + HumanMessage(content=formatted), + ] + return formatted def _get_default_template(self) -> str: """Returns the default template for the task type.""" diff --git a/coolprompt/evaluator/metrics.py b/coolprompt/evaluator/metrics.py index e1271341..d39e79b9 100644 --- a/coolprompt/evaluator/metrics.py +++ b/coolprompt/evaluator/metrics.py @@ -73,6 +73,10 @@ def _compute_raw( List[float]: List of float metrics (for each model answer). """ + outputs = [ + "none" if isinstance(o, str) and not o.strip() else o + for o in outputs + ] return [ self._postprocessing( self._metric.compute( diff --git a/coolprompt/optimizer/reflective_prompt/__init__.py b/coolprompt/optimizer/reflective_prompt/__init__.py index fd0b6dcc..31770bd8 100644 --- a/coolprompt/optimizer/reflective_prompt/__init__.py +++ b/coolprompt/optimizer/reflective_prompt/__init__.py @@ -1,3 +1,19 @@ -from coolprompt.optimizer.reflective_prompt.run import ReflectiveMethod, reflectiveprompt - -__all__ = ["reflectiveprompt", "ReflectiveMethod"] +from coolprompt.optimizer.reflective_prompt.run import ( + ReflectiveMethod, + reflectiveprompt, + coevo, + CoevoMethod, +) +from coolprompt.optimizer.reflective_prompt.factorized_evoluter import ( + FactorizedEvoluter, +) +from coolprompt.optimizer.reflective_prompt.coevo_evoluter import CoevoEvoluter + +__all__ = [ + "reflectiveprompt", + "ReflectiveMethod", + "coevo", + "CoevoMethod", + "FactorizedEvoluter", + "CoevoEvoluter", +] diff --git a/coolprompt/optimizer/reflective_prompt/coevo_base_evoluter.py b/coolprompt/optimizer/reflective_prompt/coevo_base_evoluter.py new file mode 100644 index 00000000..be3e055a --- /dev/null +++ b/coolprompt/optimizer/reflective_prompt/coevo_base_evoluter.py @@ -0,0 +1,1268 @@ +import os +import time +import yaml +from typing import Dict, List, Optional, Tuple, Any + +import numpy as np +import statistics +from scipy.special import softmax +from sklearn.metrics.pairwise import cosine_similarity + +from langchain_core.messages.ai import AIMessage +from langchain_core.language_models.base import BaseLanguageModel + +from coolprompt.evaluator import Evaluator +from coolprompt.optimizer.reflective_prompt.prompt import Prompt, PromptOrigin +from coolprompt.utils.logging_config import logger + +from coolprompt.utils.prompt_templates.reflective_templates_fixed_role import ( + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_FIXED_ROLE, + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_FIXED_ROLE, + REFLECTIVEPROMPT_MUTATION_TEMPLATE_FIXED_ROLE, + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_FIXED_ROLE, + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_FIXED_ROLE, + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_FIXED_ROLE, +) +from coolprompt.utils.prompt_templates.reflective_templates_no_role import ( + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_NO_ROLE, + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_NO_ROLE, + REFLECTIVEPROMPT_MUTATION_TEMPLATE_NO_ROLE, + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_NO_ROLE, + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_NO_ROLE, + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_NO_ROLE, +) +from coolprompt.utils.prompt_templates.reflective_templates_coevolution import ( + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO, + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO, + REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO, + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO, + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO, + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO, + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_BASE, + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_BASE, + REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_BASE, + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_3F, + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO_3F, + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_3F, + REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_3F, + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO_3F, + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_3F, +) +from coolprompt.utils.prompt_templates.reflective_templates_text_only import ( + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_TEXT_ONLY, + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_TEXT_ONLY, + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_TEXT_ONLY, + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_TEXT_ONLY, + REFLECTIVEPROMPT_MUTATION_TEMPLATE_TEXT_ONLY, + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_TEXT_ONLY, +) +from coolprompt.utils.prompt_templates.reflective_templates_factorized import ( + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_ROLE_ONLY, + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_ROLE_ONLY, + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_ROLE_ONLY, + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_ROLE_ONLY, + REFLECTIVEPROMPT_MUTATION_TEMPLATE_ROLE_ONLY, + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_CONSTRAINTS_ONLY, + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY, + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY, + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_CONSTRAINTS_ONLY, + REFLECTIVEPROMPT_MUTATION_TEMPLATE_CONSTRAINTS_ONLY, +) +from coolprompt.utils.parsing import extract_answer, extract_json + +_embedding_model = None + + +def _get_embedding_model(): + global _embedding_model + if _embedding_model is None: + from sentence_transformers import SentenceTransformer + + _embedding_model = SentenceTransformer("all-MiniLM-L6-v2") + return _embedding_model + + +class ReflectiveEvoluter: + """ + ReflectiveEvoluter class that represents evoluter for ReflectivePrompt + + Attributes: + model: langchain.BaseLanguageModel class of model to use. + evaluator: evaluator (Evaluator) to compute metrics. + train_dataset: a dataset to use while training. + train_targets: string targets for train dataset. + validation_dataset: a dataset to use while validating final prompts. + validation_targets: string targets for validation dataset. + problem_description: a string that contains + short description of problem to optimize. + initial_prompt: initial prompt to start evolution from. + Will be automatically generated if not provided. + Defaults to None. + population_size: an integer fixed size of prompt population. + Defaults to 10. + num_epochs: an integer number of epochs to evaluate. + Defaults to 10. + use_cache: a boolean variable. + Either to use caching files or not. + output_path: a path to store logs of evolution. + elitist: a prompt with highest score in population. + best_score_overall: best evaluation score during evolution. + best_prompt_overall: text of prompt with best score overall. + iteration: current iteration (epoch) of evolution. + PROMPT_TAGS: start and end tags for prompt extraction. + HINT_TAGS: start and end tags for hint extraction. + """ + + PROMPT_TAGS = ("", "") + HINT_TAGS = ("", "") + ROLE_LENGTH_ALPHA: float = 0.02 + ROLE_PROMPT_SIM_THRESHOLD: float = 0.72 + ROLE_PROMPT_SIM_ALPHA: float = 0.05 + ELITIST_MAX_FREEZE: int = 3 + BAD_EXAMPLES_TOP_K: int = 3 + PREVIEW_LEN: int = 80 + HALL_OF_FAME_MIN_SIZE: int = 10 + + def __init__( + self, + model: BaseLanguageModel, + evaluator: Evaluator, + train_dataset: List[str], + train_targets: List[str], + validation_dataset: List[str], + validation_targets: List[str], + problem_description: str, + initial_prompt: Optional[str] = None, + initial_role: Optional[str] = None, + initial_constraints: Optional[str] = None, + evolve_role: bool = True, + evolve_constraints: bool = False, + population_size: int = 10, + num_epochs: int = 10, + output_path: str = "./reflectiveprompt_outputs", + use_cache: bool = True, + use_enhancements: bool = True, + use_bad_examples: Optional[bool] = None, + freeze_text: bool = False, + text_only: bool = False, + val_evaluator: Optional[Evaluator] = None, + ) -> None: + self.model = model + self.evaluator = evaluator + self.val_evaluator = val_evaluator or evaluator + self.train_dataset = train_dataset + self.train_targets = train_targets + self.validation_dataset = validation_dataset + self.validation_targets = validation_targets + self.use_cache = use_cache + self.population_size = population_size + self.num_epochs = num_epochs + self.problem_description = problem_description + self.output_path = output_path + self.initial_prompt = initial_prompt + self.initial_role = initial_role + self.initial_constraints = initial_constraints or "" + self.evolve_role = evolve_role + self.evolve_constraints = evolve_constraints + self.use_enhancements = use_enhancements + self.use_bad_examples = ( + use_enhancements if use_bad_examples is None else use_bad_examples + ) + self.freeze_text = freeze_text + self.text_only = text_only + self._role_only = ( + self.evolve_role + and self.freeze_text + and not self.evolve_constraints + ) + self._constraints_only = ( + not self.evolve_role + and bool(self.initial_role) + and self.evolve_constraints + ) + + self.elitist = None + self._long_term_reflection_str = "" + self.best_score_overall = None + self.best_prompt_overall = None + self.best_role_overall = None + self.best_constraints_overall = None + self.iteration = 0 + self._elitist_freeze_count: int = 0 + self._prev_elitist_role: str = "" + self._hall_of_fame: List[Prompt] = [] + self._elitist_bad_examples: List[Dict] = [] + + self._setup_templates() + + def _setup_templates(self) -> None: + """Selects prompt templates based on the active evolution mode.""" + if self.text_only: + self._paraphrasing_template = ( + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_TEXT_ONLY + ) + self._crossover_template = ( + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_TEXT_ONLY + ) + self._mutation_template = ( + REFLECTIVEPROMPT_MUTATION_TEMPLATE_TEXT_ONLY + ) + self._short_term_template = ( + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_TEXT_ONLY + ) + self._long_term_template = ( + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_TEXT_ONLY + ) + self._initial_prompt_template = ( + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_TEXT_ONLY + ) + elif self._role_only: + self._paraphrasing_template = ( + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_ROLE_ONLY + ) + self._crossover_template = ( + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_ROLE_ONLY + ) + self._mutation_template = ( + REFLECTIVEPROMPT_MUTATION_TEMPLATE_ROLE_ONLY + ) + self._short_term_template = ( + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_ROLE_ONLY + ) + self._long_term_template = ( + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_ROLE_ONLY + ) + self._initial_prompt_template = ( + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO + ) + elif self._constraints_only: + self._paraphrasing_template = ( + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_CONSTRAINTS_ONLY + ) + self._crossover_template = ( + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_CONSTRAINTS_ONLY + ) + self._mutation_template = ( + REFLECTIVEPROMPT_MUTATION_TEMPLATE_CONSTRAINTS_ONLY + ) + self._short_term_template = ( + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY + ) + self._long_term_template = ( + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY + ) + self._initial_prompt_template = ( + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO + ) + elif self.evolve_role and self.evolve_constraints: + self._paraphrasing_template = ( + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO_3F + ) + self._crossover_template = ( + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_3F + ) + self._mutation_template = ( + REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_3F + ) + self._short_term_template = ( + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_3F + ) + self._long_term_template = ( + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO_3F + ) + self._initial_prompt_template = ( + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_3F + ) + elif self.evolve_role: + self._paraphrasing_template = ( + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO + ) + if self.use_enhancements: + self._crossover_template = ( + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO + ) + self._mutation_template = ( + REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO + ) + self._short_term_template = ( + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO + ) + else: + self._crossover_template = ( + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_BASE + ) + self._mutation_template = ( + REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_BASE + ) + self._short_term_template = ( + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_BASE + ) + self._long_term_template = ( + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO + ) + self._initial_prompt_template = ( + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO + ) + elif not self.initial_role: + self._paraphrasing_template = ( + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_NO_ROLE + ) + self._crossover_template = ( + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_NO_ROLE + ) + self._mutation_template = REFLECTIVEPROMPT_MUTATION_TEMPLATE_NO_ROLE + self._short_term_template = ( + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_NO_ROLE + ) + self._long_term_template = ( + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_NO_ROLE + ) + self._initial_prompt_template = ( + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_NO_ROLE + ) + else: + self._paraphrasing_template = ( + REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_FIXED_ROLE + ) + self._crossover_template = ( + REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_FIXED_ROLE + ) + self._mutation_template = ( + REFLECTIVEPROMPT_MUTATION_TEMPLATE_FIXED_ROLE + ) + self._short_term_template = ( + REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_FIXED_ROLE + ) + self._long_term_template = ( + REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_FIXED_ROLE + ) + self._initial_prompt_template = ( + REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_FIXED_ROLE + ) + + def _reranking(self, population: List[Prompt]) -> List[Prompt]: + """ + Sorts given population of prompts by their scores in descending order. + + Args: + population (List[Prompt]): population to sort. + + Returns: + List[Prompt]: sorted population. + """ + return list( + sorted(population, key=lambda prompt: prompt.score, reverse=True) + ) + + @staticmethod + def _role_prompt_sim(role: str, prompt_text: str) -> float: + """Cosine similarity between system_behavior and task_description. + Computed from all-MiniLM-L6-v2 sentence-transformer embeddings. + High similarity means the two components are redundant. + Returns 0.0 if either string is empty. + """ + if not role or not prompt_text: + return 0.0 + model = _get_embedding_model() + embs = model.encode([role, prompt_text]) + return float( + cosine_similarity(embs[0].reshape(1, -1), embs[1].reshape(1, -1))[ + 0 + ][0] + ) + + def _update_hall_of_fame(self, population: List[Prompt]) -> None: + seen = {(p.text, p.role, p.constraints) for p in self._hall_of_fame} + for p in population: + if p.score is None: + continue + key = (p.text, p.role, p.constraints) + if key not in seen: + self._hall_of_fame.append( + Prompt( + text=p.text, + role=p.role, + constraints=p.constraints, + origin=p.origin, + score=p.score, + ) + ) + seen.add(key) + self._hall_of_fame.sort(key=lambda x: x.score, reverse=True) + max_size = max(self.population_size * 2, self.HALL_OF_FAME_MIN_SIZE) + self._hall_of_fame = self._hall_of_fame[:max_size] + + def _format_bad_examples(self) -> str: + if not self.use_bad_examples or not self._elitist_bad_examples: + return "(none)" + lines = [] + for i, ex in enumerate(self._elitist_bad_examples, 1): + inp = ex.input[:120] + out = ex.output + correct = ex.correct + lines.append( + f"{i}. Input: {inp}\n Got: {out} | Expected: {correct}" + ) + return "\n".join(lines) + + def _format_top_prompts_history(self, top_k: int = 5) -> str: + if not self._hall_of_fame: + return "(none)" + entries = self._hall_of_fame[:top_k] + lines = [] + for i, p in enumerate(entries, 1): + score_str = self._format_score(p.score) + if self._role_only: + content = p.role or "(empty)" + lines.append(f"{i}. [score={score_str}] {content[:120]}") + elif self._constraints_only: + content = p.constraints or "(empty)" + lines.append(f"{i}. [score={score_str}] {content[:120]}") + elif self.evolve_role and self.evolve_constraints: + role = (p.role or "(empty)")[: self.PREVIEW_LEN] + text = (p.text or "(empty)")[: self.PREVIEW_LEN] + constraints = (p.constraints or "(empty)")[: self.PREVIEW_LEN] + lines.append( + f"{i}. [score={score_str}]\n system_behavior: {role}\n task_description: {text}\n output_constraints: {constraints}" + ) + elif self.evolve_role: + role = (p.role or "(empty)")[: self.PREVIEW_LEN] + text = (p.text or "(empty)")[: self.PREVIEW_LEN] + lines.append( + f"{i}. [score={score_str}]\n system_behavior: {role}\n task_description: {text}" + ) + else: + content = p.text or "(empty)" + lines.append(f"{i}. [score={score_str}] {content[:120]}") + return "\n".join(lines) + + def _aggregate_bad_examples( + self, population: List[Prompt], top_k: int = BAD_EXAMPLES_TOP_K + ) -> None: + scored = [ + p for p in population if p.score is not None and p.bad_examples + ] + if not scored: + return + scored.sort(key=lambda p: p.score, reverse=True) + top_half = scored[: max(1, len(scored) // 2)] + counts: Dict[str, Dict] = {} + for p in top_half: + for ex in p.bad_examples: + key = ex.input + if key not in counts: + counts[key] = {"count": 0, "ex": ex} + counts[key]["count"] += 1 + sorted_examples = sorted( + counts.values(), key=lambda x: x["count"], reverse=True + ) + self._elitist_bad_examples = [x["ex"] for x in sorted_examples[:top_k]] + + def _format_score(self, score) -> str: + if not self.use_enhancements or score is None: + return "N/A" + return f"{score:.4f}" + + def _evaluate(self, prompt: Prompt, split="train") -> None: + """Evaluates given prompt on self.dataset and records the score. + When evolve_role=True and split=='train': + - A length penalty proportional to role length is subtracted, + discouraging bloated roles when scores are close. + + Args: + prompt (Prompt): a prompt to evaluate. + split (str, optional): Which split of dataset to use. + Defaults to 'train'. + """ + if split == "train": + dataset, targets = self.train_dataset, self.train_targets + else: + dataset, targets = self.validation_dataset, self.validation_targets + + eval_role = prompt.role + ev = self.evaluator if split == "train" else self.val_evaluator + result = ev.evaluate( + prompt=prompt.text, + dataset=dataset, + targets=targets, + system_role=eval_role if eval_role else None, + constraints=prompt.constraints if self.evolve_constraints else None, + failed_examples=( + 10 if split == "train" and self.use_bad_examples else None + ), + ) + if isinstance(result, tuple): + score, bad_examples = result + prompt.set_bad_examples(bad_examples) + else: + score = result + + if self.evolve_role and split == "train": + if self.use_enhancements: + score = score - self.ROLE_LENGTH_ALPHA * len(prompt.role) / 1000 + + if prompt.role: + sim = self._role_prompt_sim(prompt.role, prompt.text) + if sim > self.ROLE_PROMPT_SIM_THRESHOLD: + score -= self.ROLE_PROMPT_SIM_ALPHA * ( + sim - self.ROLE_PROMPT_SIM_THRESHOLD + ) + + prompt.set_score(score) + + def _evaluation( + self, population: List[Prompt], split: str = "train" + ) -> None: + """Evaluation operation for prompts population. + Evaluates every prompt in population and records the results. + + Args: + population (List[Prompt]): population of prompts to evaluate. + split (str, optional): Which split of dataset to use. + Defaults to 'train'. + """ + logger.info("Evaluating population...") + for prompt in population: + self._evaluate(prompt, split=split) + if split == "train": + self._aggregate_bad_examples(population) + + def _create_initial_prompt(self) -> Tuple[str, str, str]: + """Creates an initial prompt according to provided problem description + + Returns: + Tuple[str, str, str]: initial prompt + """ + request = self._initial_prompt_template.format( + PROBLEM_DESCRIPTION=self.problem_description + ) + answer = self._llm_query([request])[0] + extracted = extract_json(answer) + if extracted is None: + extracted = {} + + if self.evolve_role: + role = extracted.get("system_behavior", extracted.get("role", "")) + else: + role = self.initial_role or "" + + prompt = extracted.get( + "task_description", + extracted.get( + "prompt", + extract_answer( + answer, self.PROMPT_TAGS, format_mismatch_label="" + ), + ), + ) + constraints = ( + extracted.get("output_constraints", "") + if self.evolve_constraints + else "" + ) + return role, prompt, constraints + + def _init_pop(self) -> List[Prompt]: + """Creates initial population of prompts. + + Returns: + List[Prompt]: initial population. + """ + + logger.info("Initializing population...") + if self.initial_prompt is None: + generated_role, self.initial_prompt, generated_constraints = ( + self._create_initial_prompt() + ) + if self.evolve_role and not self.initial_role: + self.initial_role = generated_role + if self.evolve_constraints and not self.initial_constraints: + self.initial_constraints = generated_constraints + + if self.initial_role is None: + self.initial_role = "" + + fmt_kwargs = { + "ROLE": self.initial_role, + "PROMPT": self.initial_prompt, + "NUM_PROMPTS": self.population_size, + "PROBLEM_DESCRIPTION": self.problem_description, + } + if self.evolve_constraints: + fmt_kwargs["CONSTRAINTS"] = self.initial_constraints + request = self._paraphrasing_template.format(**fmt_kwargs) + answer = self._llm_query([request])[0] + extracted = extract_json(answer) + if extracted is None or "prompts" not in extracted: + logger.warning( + "Failed to extract prompts from LLM response, using fallback" + ) + prompts_data = [ + {"role": self.initial_role, "prompt": self.initial_prompt} + ] * self.population_size + else: + prompts_data = extracted["prompts"] + + if not isinstance(prompts_data, list) or len(prompts_data) == 0: + logger.warning("Invalid prompts_data format, using fallback") + prompts_data = [ + {"role": self.initial_role, "prompt": self.initial_prompt} + ] * self.population_size + + initial_population = [] + fixed_role = self.initial_role if not self.evolve_role else None + + for p_data in prompts_data: + if isinstance(p_data, dict): + if self.evolve_role: + role = p_data.get( + "system_behavior", + p_data.get("role", self.initial_role), + ) + else: + role = fixed_role or "" + text = p_data.get( + "task_description", + p_data.get("prompt", str(p_data)), + ) + constraints = ( + p_data.get("output_constraints", "") + if self.evolve_constraints + else "" + ) + if self._role_only or self._constraints_only: + text = self.initial_prompt + initial_population.append( + Prompt( + text=text, + role=role, + constraints=constraints, + origin=PromptOrigin.APE, + ) + ) + else: + role = ( + fixed_role or self.initial_role + if not self.evolve_role + else self.initial_role + ) + initial_population.append( + Prompt(text=p_data, role=role, origin=PromptOrigin.APE) + ) + + initial_population[-1] = Prompt( + text=self.initial_prompt, + role=( + fixed_role or self.initial_role + if not self.evolve_role + else self.initial_role + ), + constraints=( + self.initial_constraints if self.evolve_constraints else "" + ), + origin=PromptOrigin.MANUAL, + ) + self._evaluation(initial_population) + initial_population = self._reranking(initial_population) + return initial_population + + def _cache_data(self, data: Any, savepath: os.PathLike) -> None: + """Writes the data to the yaml file. + + Args: + data (Any): data to be cached. + savepath (os.PathLike): a path to saving file. + """ + os.makedirs(os.path.dirname(savepath), exist_ok=True) + with open(savepath, "w") as f: + yaml.dump(data, f) + + def _cache_population( + self, population: List[Prompt], savepath: os.PathLike + ) -> None: + """Caching a population of prompts to file. + If self.use_cache is False this function will do nothing. + + Args: + population (List[Prompt]): prompt population. + savepath (os.PathLike): a path to saving file. + """ + if self.use_cache is False: + return + + best_score = population[0].score + average_score = statistics.mean([prompt.score for prompt in population]) + data = { + "best_score": best_score, + "average_score": average_score, + "prompts": [prompt.to_dict() for prompt in population], + } + self._cache_data(data, savepath) + + def _selection(self, population: List[Prompt]) -> List[Prompt]: + """Provides selection operation. + In current implementation we want to select parents + with different scores. + But when there is difficult to do so (trial number check), + it will just sample anyways. + + Probabilities - normalized scores. + + Args: + population (List[Prompt]): prompt population to select from. + + Returns: + List[Prompt]: selected prompts. + """ + selected_population = [] + + scores = np.array([prompt.score for prompt in population]) + scores = np.clip(scores, 0, None) + if np.sum(scores) == 0: + probas = np.ones(len(scores)) / len(scores) + else: + probas = scores / np.sum(scores) + + trial = 0 + anyways = False + while len(selected_population) < 2 * self.population_size: + parents = np.random.choice( + population, size=2, replace=False, p=probas + ) + if parents[0].score != parents[1].score or anyways: + selected_population.extend(parents) + trial += 1 + if trial > 1000: + anyways = True + + return selected_population + + def _survive( + self, population: List[Prompt], temperature: float = None + ) -> List[Prompt]: + """Final selection before going into new epoch. + Probabilities are based on softmax function with temperature (if set). + + Args: + population (List[Prompt]): population to select from. + temperature (float, optional): temperature parameter for softmax. + Defaults to None. + + Returns: + List[Prompt]: selected (survived) prompts. + """ + scores = np.array([prompt.score for prompt in population]) + if temperature is not None: + scores /= temperature + probas = softmax(scores) + return np.random.choice( + population, size=self.population_size, replace=False, p=probas + ) + + def _gen_short_term_reflection_prompt( + self, prompt1: Prompt, prompt2: Prompt + ) -> Tuple[str, Prompt, Prompt]: + """Generates short-term reflection request into model. + + Args: + prompt1 (Prompt): first prompt. + prompt2 (Prompt): second prompt. + + Returns: + Tuple[str, Prompt, Prompt]: + string request, worse prompt, better prompt. + """ + if prompt1.score > prompt2.score: + better_prompt, worse_prompt = prompt1, prompt2 + else: + better_prompt, worse_prompt = prompt2, prompt1 + + fmt_kwargs = { + "PROBLEM_DESCRIPTION": self.problem_description, + "WORSE_PROMPT_ROLE": worse_prompt.role, + "WORSE_PROMPT_TEXT": worse_prompt.text, + "BETTER_PROMPT_ROLE": better_prompt.role, + "BETTER_PROMPT_TEXT": better_prompt.text, + "WORSE_SCORE": self._format_score(worse_prompt.score), + "BETTER_SCORE": self._format_score(better_prompt.score), + } + if self.evolve_constraints or self._constraints_only: + fmt_kwargs["WORSE_PROMPT_CONSTRAINTS"] = worse_prompt.constraints + fmt_kwargs["BETTER_PROMPT_CONSTRAINTS"] = better_prompt.constraints + if self._role_only or self._constraints_only: + fmt_kwargs["FROZEN_PROMPT_TEXT"] = self.initial_prompt + fmt_kwargs["FROZEN_PROMPT_ROLE"] = self.initial_role or "" + request = self._short_term_template.format(**fmt_kwargs) + + return request, worse_prompt, better_prompt + + def _make_output_path(self, filename: str) -> os.PathLike: + """Creates full path for logging based on current iteration. + + Args: + filename (str): the file name to save. + + Returns: + os.PathLike: final path to save. + """ + return os.path.join( + self.output_path, f"Iteration{self.iteration}", f"{filename}.yaml" + ) + + def _short_term_reflection( + self, + population: list[Prompt], + ) -> Tuple[List[str], List[Prompt], List[Prompt]]: + """Short-term reflection before crossovering two individuals. + + Args: + population (list[Prompt]): parenting population. + + Returns: + Tuple[List[str], List[Prompt], List[Prompt]]: + generated short-term hints, + worse prompts, + better prompts. + """ + requests = [] + worse_prompts = [] + better_prompts = [] + for i in range(0, len(population), 2): + parent_1 = population[i] + parent_2 = population[i + 1] + + request, worse_p, better_p = self._gen_short_term_reflection_prompt( + parent_1, parent_2 + ) + requests.append(request) + worse_prompts.append(worse_p) + better_prompts.append(better_p) + + responses = self._llm_query(requests) + responses = [ + extract_answer(response, self.HINT_TAGS, format_mismatch_label="") + for response in responses + ] + return responses, worse_prompts, better_prompts + + def _crossover( + self, + short_term_reflection_tuple: Tuple[ + List[str], List[Prompt], List[Prompt] + ], + ) -> List[Prompt]: + """Provides crossover operation. + + Args: + short_term_reflection_tuple + (Tuple[List[str], List[Prompt], List[Prompt]]): + outputs of short-term reflection. + + Returns: + List[Prompt]: new crossed prompts population. + """ + reflection_contents, worse_prompts, better_prompts = ( + short_term_reflection_tuple + ) + requests = [] + for reflection, worse_p, better_p in zip( + reflection_contents, worse_prompts, better_prompts + ): + fmt_kwargs = { + "PROBLEM_DESCRIPTION": self.problem_description, + "WORSE_PROMPT_ROLE": worse_p.role, + "WORSE_PROMPT_TEXT": worse_p.text, + "BETTER_PROMPT_ROLE": better_p.role, + "BETTER_PROMPT_TEXT": better_p.text, + "SHORT_TERM_REFLECTION": reflection, + "WORSE_SCORE": self._format_score(worse_p.score), + "BETTER_SCORE": self._format_score(better_p.score), + } + if self.evolve_constraints or self._constraints_only: + fmt_kwargs["WORSE_PROMPT_CONSTRAINTS"] = worse_p.constraints + fmt_kwargs["BETTER_PROMPT_CONSTRAINTS"] = better_p.constraints + if self._role_only or self._constraints_only: + fmt_kwargs["FROZEN_PROMPT_TEXT"] = self.initial_prompt + fmt_kwargs["FROZEN_PROMPT_ROLE"] = self.initial_role or "" + request = self._crossover_template.format(**fmt_kwargs) + requests.append(request) + + responses = self._llm_query(requests) + crossed_population = [] + for i, response in enumerate(responses): + extracted = extract_json(response) + if extracted is None: + extracted = {} + + if self._role_only: + role = extracted.get( + "system_behavior", extracted.get("role", "") + ) + text = self.initial_prompt + constraints = "" + elif self._constraints_only: + role = self.initial_role or "" + text = self.initial_prompt + constraints = extracted.get("output_constraints", "") + else: + if self.evolve_role: + role = extracted.get( + "system_behavior", extracted.get("role", "") + ) + else: + better_p = better_prompts[i] + role = ( + better_p.role + if better_p.role + else self.initial_role or "" + ) + text = extracted.get( + "task_description", + extracted.get( + "prompt", + extract_answer( + response, + self.PROMPT_TAGS, + format_mismatch_label="", + ), + ), + ) + constraints = ( + extracted.get("output_constraints", "") + if self.evolve_constraints + else "" + ) + crossed_population.append( + Prompt(text=text, role=role, constraints=constraints) + ) + + assert len(crossed_population) == self.population_size + return crossed_population + + def _update_elitist(self, population: List[Prompt]) -> None: + scores = [prompt.score for prompt in population] + best_score, best_sample_idx = max(scores), np.argmax(np.array(scores)) + + if ( + self.best_score_overall is None + or best_score >= self.best_score_overall + ): + self.best_score_overall = best_score + self.best_prompt_overall = population[best_sample_idx].text + self.best_constraints_overall = population[ + best_sample_idx + ].constraints + self.elitist = population[best_sample_idx] + logger.info(f"""Iteration {self.iteration} + Elitist score: {self.best_score_overall}""") + logger.debug(f"Elitist text:\n{self.elitist.text}") + + def _update_iter(self, population: List[Prompt]) -> None: + """Updates iteration. Cache current state. + Also tracks elitist freeze: if the elitist role has not changed + for ELITIST_MAX_FREEZE consecutive epochs, forces the best + candidate with a different role to become the new elitist. + + Args: + population (List[Prompt]): current population. + """ + logger.info(f"Iteration {self.iteration} finished...") + logger.info(f"Best score: {self.best_score_overall}") + + if self.use_enhancements: + current_role = self.elitist.role if self.elitist else "" + if current_role == self._prev_elitist_role: + self._elitist_freeze_count += 1 + else: + self._elitist_freeze_count = 0 + self._prev_elitist_role = current_role + + if self._elitist_freeze_count >= self.ELITIST_MAX_FREEZE: + diverse = [ + p + for p in population + if p.role != current_role and p.score is not None + ] + if diverse: + best_diverse = max(diverse, key=lambda p: p.score) + logger.debug( + f"Elitist frozen {self._elitist_freeze_count} epochs, " + f"forcing diverse candidate: '{best_diverse.role[:60]}'" + ) + self.elitist = best_diverse + self._prev_elitist_role = best_diverse.role + self._elitist_freeze_count = 0 + + population = self._reranking(population) + self._cache_population(population, self._make_output_path("population")) + + self.iteration += 1 + + def _long_term_reflection(self, short_term_reflections: List[str]) -> None: + """Long-term reflection before mutation. + + Args: + short_term_reflections (List[str]): short-term reflections. + """ + long_term_kwargs = dict( + PROBLEM_DESCRIPTION=self.problem_description, + PRIOR_LONG_TERM_REFLECTION=self._long_term_reflection_str, + NEW_SHORT_TERM_REFLECTIONS="\n".join(short_term_reflections), + ) + if ( + self._role_only + or self._constraints_only + or self.text_only + or self.evolve_role + ): + long_term_kwargs["TOP_PROMPTS_HISTORY"] = ( + self._format_top_prompts_history() + ) + if self._constraints_only: + long_term_kwargs["FROZEN_PROMPT_TEXT"] = self.initial_prompt + long_term_kwargs["FROZEN_PROMPT_ROLE"] = self.initial_role or "" + request = self._long_term_template.format(**long_term_kwargs) + + response = self._llm_query([request])[0] + + self._long_term_reflection_str = extract_answer( + response, self.HINT_TAGS, format_mismatch_label="" + ) + + def _llm_query(self, requests: List[str]) -> List[str]: + """Provides api to query requests to the model. + Retries up to 3 times with exponential backoff on failure. + + Args: + requests (List[str]): string requests. + + Returns: + List[str]: model answers. + """ + for attempt in range(3): + try: + answers = self.model.batch(requests) + return [ + a.content if isinstance(a, AIMessage) else a + for a in answers + ] + except Exception as e: + if attempt < 2: + logger.warning( + f"LLM query failed (attempt {attempt + 1}): {e}. Retrying..." + ) + time.sleep(5 * (attempt + 1)) + else: + raise + + def _mutate(self) -> List[Prompt]: + """Elitist-based mutation. + + Returns: + List[Prompt]: generated population. + """ + fmt_kwargs = { + "PROBLEM_DESCRIPTION": self.problem_description, + "LONG_TERM_REFLECTION": self._long_term_reflection_str, + "ELITIST_PROMPT_ROLE": self.elitist.role, + "ELITIST_PROMPT_TEXT": self.elitist.text, + "ELITIST_SCORE": self._format_score(self.elitist.score), + } + if self.evolve_constraints or self._constraints_only: + fmt_kwargs["ELITIST_PROMPT_CONSTRAINTS"] = self.elitist.constraints + if ( + self._role_only + or self._constraints_only + or self.text_only + or self.evolve_role + ): + fmt_kwargs["BAD_EXAMPLES"] = self._format_bad_examples() + if self._role_only or self._constraints_only: + fmt_kwargs["FROZEN_PROMPT_TEXT"] = self.initial_prompt + fmt_kwargs["FROZEN_PROMPT_ROLE"] = self.initial_role or "" + request = self._mutation_template.format(**fmt_kwargs) + responses = self._llm_query([request] * self.population_size) + mutated_population = [] + fixed_role = ( + self.elitist.role if self.elitist and not self.evolve_role else None + ) + if fixed_role is None and not self.evolve_role: + fixed_role = self.initial_role or "" + + for response in responses: + extracted = extract_json(response) + if extracted is None: + extracted = {} + + if self._role_only: + role = extracted.get( + "system_behavior", extracted.get("role", "") + ) + text = self.initial_prompt + constraints = "" + elif self._constraints_only: + role = self.initial_role or "" + text = self.initial_prompt + constraints = extracted.get("output_constraints", "") + else: + if self.evolve_role: + role = extracted.get( + "system_behavior", extracted.get("role", "") + ) + else: + role = fixed_role + text = extracted.get( + "task_description", + extracted.get( + "prompt", + extract_answer( + response, + self.PROMPT_TAGS, + format_mismatch_label="", + ), + ), + ) + constraints = ( + extracted.get("output_constraints", "") + if self.evolve_constraints + else "" + ) + mutated_population.append( + Prompt( + text=text, + role=role, + constraints=constraints, + origin=PromptOrigin.MUTATED, + ) + ) + return mutated_population + + def evolution(self, skip_validation: bool = False) -> str: + """Provides evolution operation. + + Selection -> Short-term reflection -> Long-term reflection + -> Elitist-based mutation -> Survival. + + After all self.num_epochs epochs the best three prompts are selected. + They will be evaluated on test split of dataset then. + And based on their test scores, + the best prompt will be returned. + + Returns: + str: best evoluted prompt + """ + + population = np.array(self._init_pop()) + self._cache_population( + population, self._make_output_path("initial_population.yaml") + ) + + while self.iteration < self.num_epochs: + parent_population = self._selection(population) + + short_term_reflection_tuple = self._short_term_reflection( + parent_population + ) + self._cache_data( + short_term_reflection_tuple[0], + self._make_output_path("short_term_reflections"), + ) + + crossed_population = self._crossover(short_term_reflection_tuple) + + self._evaluation(crossed_population) + self._update_elitist(crossed_population) + + self._long_term_reflection(short_term_reflection_tuple[0]) + self._cache_data( + self._long_term_reflection_str, + self._make_output_path("long_term_reflection"), + ) + + mutated_population = self._mutate() + self._evaluation(mutated_population) + + population = np.append(population, np.array(crossed_population)) + population = np.append(population, np.array(mutated_population)) + self._update_elitist(population) + population = self._survive(population, temperature=1e-1) + + if self.elitist is not None and self.elitist not in population: + logger.debug("Elitist should always live") + population = np.append(population, np.array([self.elitist])) + + if self.use_enhancements: + self._update_hall_of_fame(population) + self._cache_data( + self._elitist_bad_examples, + self._make_output_path("bad_examples"), + ) + self._cache_data( + [ + { + "score": self._format_score(p.score), + "text": p.text, + "role": p.role, + "constraints": p.constraints, + } + for p in self._hall_of_fame[:5] + ], + self._make_output_path("top_prompts_history"), + ) + self._update_iter(population) + + logger.info(f"BEST TRAIN SCORE: {self.best_score_overall}") + + population = self._reranking(population) + final_candidates = list(population[:3]) + if self.elitist is not None: + if not any( + c.text == self.elitist.text + and c.role == self.elitist.role + and c.constraints == self.elitist.constraints + for c in final_candidates + ): + final_candidates.append(self.elitist) + + if self.use_enhancements: + seen = {(c.text, c.role, c.constraints) for c in final_candidates} + for hof_p in self._hall_of_fame: + if (hof_p.text, hof_p.role, hof_p.constraints) not in seen: + final_candidates.append(hof_p) + seen.add((hof_p.text, hof_p.role, hof_p.constraints)) + if len(final_candidates) >= 6: + break + + if not skip_validation: + logger.info( + f"Final validation: {len(final_candidates)} candidates " + f"({'with HoF' if self.use_enhancements else 'no HoF'})" + ) + final_candidates = np.array(final_candidates) + self._evaluation(final_candidates, split="validation") + final_candidates = self._reranking(final_candidates) + self._cache_population( + final_candidates, + self._make_output_path("best_prompts_infer.yaml"), + ) + self.elitist = final_candidates[0] + self.best_prompt_overall = self.elitist.text + self.best_role_overall = self.elitist.role + self.best_constraints_overall = self.elitist.constraints + self.best_score_overall = self.elitist.score + logger.info(f"BEST VALIDATION SCORE: {self.best_score_overall}") + logger.debug(f"BEST ROLE:\n{self.best_role_overall}") + logger.debug(f"BEST PROMPT:\n{self.best_prompt_overall}") + if self.best_constraints_overall: + logger.debug( + f"BEST CONSTRAINTS:\n{self.best_constraints_overall}" + ) + else: + logger.info("Skipping final validation (intermediate phase).") + if self.elitist is not None: + self.best_prompt_overall = self.elitist.text + self.best_role_overall = self.elitist.role + self.best_constraints_overall = self.elitist.constraints + logger.info(f"BEST TRAIN SCORE (kept): {self.best_score_overall}") + + return self.best_prompt_overall diff --git a/coolprompt/optimizer/reflective_prompt/coevo_evoluter.py b/coolprompt/optimizer/reflective_prompt/coevo_evoluter.py new file mode 100644 index 00000000..f836954b --- /dev/null +++ b/coolprompt/optimizer/reflective_prompt/coevo_evoluter.py @@ -0,0 +1,520 @@ +import re +from typing import Dict, List, Optional, Tuple + +from pydantic import ( + BaseModel, + ValidationError, + field_validator, + model_validator, +) +from langchain_core.language_models.base import BaseLanguageModel + +from coolprompt.evaluator import Evaluator +from coolprompt.optimizer.reflective_prompt.coevo_base_evoluter import ( + ReflectiveEvoluter, +) +from coolprompt.optimizer.reflective_prompt.prompt import Prompt, PromptOrigin +from coolprompt.utils.logging_config import logger +from coolprompt.utils.parsing import extract_json, extract_answer +from coolprompt.utils.prompt_templates.reflective_templates_coevo_enhanced import ( + PARAPHRASING_TEMPLATE_COEVO_ENH, + SHORT_TERM_REFLECTION_TEMPLATE_COEVO_ENH, + LONG_TERM_REFLECTION_TEMPLATE_COEVO_ENH, + CROSSOVER_TEMPLATE_COEVO_ENH, + MUTATION_TEMPLATE_COEVO_ENH, + PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_ENH, +) +from coolprompt.utils.prompt_templates.reflective_templates_coevo_per_field import ( + PARAPHRASING_TEMPLATE_COEVO_PF, + SHORT_TERM_REFLECTION_TEMPLATE_COEVO_PF, + LONG_TERM_REFLECTION_TEMPLATE_COEVO_PF, + CROSSOVER_TEMPLATE_COEVO_PF, + MUTATION_TEMPLATE_COEVO_PF, + PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_PF, +) + + +def _sanitize(value: str) -> str: + value = value.strip().strip('"').strip("'").strip() + value = re.sub(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f\u2028\u2029]", "", value) + return value + + +class _ThreeFieldOutput(BaseModel): + task_description: str = "" + system_behavior: str = "" + output_constraints: str = "" + + @field_validator( + "task_description", + "system_behavior", + "output_constraints", + mode="before", + ) + @classmethod + def clean_field(cls, v): + return _sanitize(str(v)) if v else "" + + @model_validator(mode="after") + def check_task_not_empty(self): + if not self.task_description: + raise ValueError("task_description is empty") + return self + + +class CoevoEvoluter(ReflectiveEvoluter): + """Evoluter that coevolves all three prompt fields simultaneously. + + Optimizes task_description, system_behavior and output_constraints + together in each epoch. Uses Pydantic to validate LLM outputs and + runs field ablation at the end to pick the best field combination. + """ + + def __init__( + self, + model: BaseLanguageModel, + evaluator: Evaluator, + train_dataset: List[str], + train_targets: List[str], + validation_dataset: List[str], + validation_targets: List[str], + problem_description: str, + initial_prompt: Optional[str] = None, + initial_role: Optional[str] = None, + initial_constraints: Optional[str] = None, + population_size: int = 10, + num_epochs: int = 10, + output_path: str = "./coevo_outputs", + use_cache: bool = True, + use_enhancements: bool = True, + use_bad_examples: Optional[bool] = None, + val_evaluator: Optional[Evaluator] = None, + ) -> None: + super().__init__( + model=model, + evaluator=evaluator, + train_dataset=train_dataset, + train_targets=train_targets, + validation_dataset=validation_dataset, + validation_targets=validation_targets, + problem_description=problem_description, + initial_prompt=initial_prompt, + initial_role=initial_role, + initial_constraints=initial_constraints, + evolve_role=True, + evolve_constraints=True, + population_size=population_size, + num_epochs=num_epochs, + output_path=output_path, + use_cache=use_cache, + use_enhancements=use_enhancements, + use_bad_examples=use_bad_examples, + freeze_text=False, + text_only=False, + val_evaluator=val_evaluator, + ) + self.candidates: List[Dict] = [] + + self._paraphrasing_template = PARAPHRASING_TEMPLATE_COEVO_ENH + self._crossover_template = CROSSOVER_TEMPLATE_COEVO_ENH + self._mutation_template = MUTATION_TEMPLATE_COEVO_ENH + self._short_term_template = SHORT_TERM_REFLECTION_TEMPLATE_COEVO_ENH + self._long_term_template = LONG_TERM_REFLECTION_TEMPLATE_COEVO_ENH + self._initial_prompt_template = PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_ENH + + def _llm_query(self, requests: List[str]) -> List[str]: + results = [] + for req in requests: + results.extend(super()._llm_query([_sanitize(req)])) + return results + + def _parse_3f_response( + self, + response: str, + fallback_text: str = "", + fallback_role: str = "", + fallback_constraints: str = "", + ) -> Dict[str, str]: + raw = extract_json(response) or {} + + try: + parsed = _ThreeFieldOutput( + task_description=raw.get("task_description") or fallback_text, + system_behavior=raw.get("system_behavior") or fallback_role, + output_constraints=raw.get("output_constraints") + or fallback_constraints, + ) + except (ValidationError, ValueError) as e: + logger.warning( + f"_parse_3f_response validation failed ({e}), using fallback" + ) + return { + "task_description": _sanitize(fallback_text), + "system_behavior": _sanitize(fallback_role), + "output_constraints": _sanitize(fallback_constraints), + } + return { + "task_description": parsed.task_description, + "system_behavior": parsed.system_behavior, + "output_constraints": parsed.output_constraints, + } + + def _crossover( + self, + short_term_reflection_tuple: Tuple[ + List[str], List[Prompt], List[Prompt] + ], + ) -> List[Prompt]: + reflection_contents, worse_prompts, better_prompts = ( + short_term_reflection_tuple + ) + requests = [] + for reflection, worse_p, better_p in zip( + reflection_contents, worse_prompts, better_prompts + ): + request = self._crossover_template.format( + PROBLEM_DESCRIPTION=self.problem_description, + WORSE_PROMPT_TEXT=worse_p.text, + WORSE_PROMPT_ROLE=worse_p.role, + WORSE_PROMPT_CONSTRAINTS=worse_p.constraints, + BETTER_PROMPT_TEXT=better_p.text, + BETTER_PROMPT_ROLE=better_p.role, + BETTER_PROMPT_CONSTRAINTS=better_p.constraints, + SHORT_TERM_REFLECTION=reflection, + WORSE_SCORE=self._format_score(worse_p.score), + BETTER_SCORE=self._format_score(better_p.score), + ) + requests.append(request) + + responses = self._llm_query(requests) + crossed_population = [] + for i, response in enumerate(responses): + fields = self._parse_3f_response( + response, + fallback_text=better_prompts[i].text, + fallback_role=better_prompts[i].role, + fallback_constraints=better_prompts[i].constraints, + ) + crossed_population.append( + Prompt( + text=fields["task_description"], + role=fields["system_behavior"], + constraints=fields["output_constraints"], + origin=PromptOrigin.EVOLUTED, + ) + ) + + assert len(crossed_population) == self.population_size + return crossed_population + + def _mutate(self) -> List[Prompt]: + request = self._mutation_template.format( + PROBLEM_DESCRIPTION=self.problem_description, + LONG_TERM_REFLECTION=self._long_term_reflection_str, + ELITIST_PROMPT_TEXT=self.elitist.text, + ELITIST_PROMPT_ROLE=self.elitist.role, + ELITIST_PROMPT_CONSTRAINTS=self.elitist.constraints, + ELITIST_SCORE=self._format_score(self.elitist.score), + BAD_EXAMPLES=self._format_bad_examples(), + ) + responses = self._llm_query([request] * self.population_size) + mutated_population = [] + for response in responses: + fields = self._parse_3f_response( + response, + fallback_text=self.elitist.text, + fallback_role=self.elitist.role, + fallback_constraints=self.elitist.constraints, + ) + mutated_population.append( + Prompt( + text=fields["task_description"], + role=fields["system_behavior"], + constraints=fields["output_constraints"], + origin=PromptOrigin.MUTATED, + ) + ) + return mutated_population + + def _eval_val(self, prompt: str, role: str, constraints: str) -> float: + result = self.val_evaluator.evaluate( + prompt=prompt, + dataset=self.validation_dataset, + targets=self.validation_targets, + system_role=role or None, + constraints=constraints or None, + ) + assert isinstance(result, float) + return result + + def _field_ablation( + self, + best_text: str, + best_role: str, + best_constraints: str, + score_text_role_constraints: Optional[float] = None, + ) -> Tuple[List[Dict], Dict]: + logger.info( + "[Field Ablation] Evaluating field combinations on validation set..." + ) + score_a = self._eval_val(best_text, "", "") + logger.info(f" text_only: {score_a:.4f}") + score_b = self._eval_val(best_text, best_role, "") + logger.info(f" text_role: {score_b:.4f}") + + candidates = [ + { + "combo": "text_only", + "prompt": best_text, + "role": "", + "constraints": "", + "val_score": score_a, + }, + { + "combo": "text_role", + "prompt": best_text, + "role": best_role, + "constraints": "", + "val_score": score_b, + }, + ] + + if best_constraints: + if score_text_role_constraints is None: + score_text_role_constraints = self._eval_val( + best_text, best_role, best_constraints + ) + logger.info( + f" text_role_constraints: {score_text_role_constraints:.4f}" + ) + candidates.append( + { + "combo": "text_role_constraints", + "prompt": best_text, + "role": best_role, + "constraints": best_constraints, + "val_score": score_text_role_constraints, + } + ) + + _combo_order = { + "text_only": 0, + "text_role": 1, + "text_role_constraints": 2, + } + best_c = max( + candidates, key=lambda c: (c["val_score"], _combo_order[c["combo"]]) + ) + logger.info( + f"Best combo: {best_c['combo']} (val={best_c['val_score']:.4f})" + ) + return candidates, best_c + + def evolution(self) -> Optional[str]: + super().evolution() + + if self.best_prompt_overall: + candidates, best_c = self._field_ablation( + best_text=self.best_prompt_overall, + best_role=self.best_role_overall or "", + best_constraints=self.best_constraints_overall or "", + score_text_role_constraints=self.best_score_overall, + ) + self.candidates = candidates + self.best_prompt_overall = best_c["prompt"] + self.best_role_overall = best_c["role"] + self.best_constraints_overall = best_c["constraints"] + self.best_score_overall = best_c["val_score"] + logger.info( + f"Field ablation done. Best combo: {best_c['combo']} " + f"(val={best_c['val_score']:.4f})" + ) + + return self.best_prompt_overall + + +class PerFieldCoevoEvoluter(CoevoEvoluter): + """CoevoEvoluter variant that uses per-field reflection hints. + + Each reflection step produces three separate hints β€” one each for + task_description, system_behavior, and output_constraints β€” instead of a + single combined hint. Crossover and mutation templates consume these + field-specific hints directly. + + All other enhancements (HoF, role length penalty, similarity penalty, + field ablation) are inherited from CoevoEvoluter unchanged. + """ + + HINT_TASK_TAGS = ("", "") + HINT_ROLE_TAGS = ("", "") + HINT_CONSTRAINTS_TAGS = ("", "") + + _FALLBACK_HINT = "(no hint)" + + def __init__(self, *args, **kwargs): + super().__init__(*args, **kwargs) + self._paraphrasing_template = PARAPHRASING_TEMPLATE_COEVO_PF + self._crossover_template = CROSSOVER_TEMPLATE_COEVO_PF + self._mutation_template = MUTATION_TEMPLATE_COEVO_PF + self._short_term_template = SHORT_TERM_REFLECTION_TEMPLATE_COEVO_PF + self._long_term_template = LONG_TERM_REFLECTION_TEMPLATE_COEVO_PF + self._initial_prompt_template = PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_PF + + self._per_field_short_hints: List[Dict[str, str]] = [] + self._per_field_long_hints: Dict[str, str] = { + "task": self._FALLBACK_HINT, + "role": self._FALLBACK_HINT, + "constraints": self._FALLBACK_HINT, + } + + def _parse_per_field_hints(self, response: str) -> Dict[str, str]: + task = extract_answer( + response, self.HINT_TASK_TAGS, format_mismatch_label="" + ).strip() + role = extract_answer( + response, self.HINT_ROLE_TAGS, format_mismatch_label="" + ).strip() + constraints = extract_answer( + response, self.HINT_CONSTRAINTS_TAGS, format_mismatch_label="" + ).strip() + return { + "task": task or self._FALLBACK_HINT, + "role": role or self._FALLBACK_HINT, + "constraints": constraints or self._FALLBACK_HINT, + } + + def _short_term_reflection(self, population): + requests = [] + worse_prompts = [] + better_prompts = [] + for i in range(0, len(population), 2): + parent_1 = population[i] + parent_2 = population[i + 1] + request, worse_p, better_p = self._gen_short_term_reflection_prompt( + parent_1, parent_2 + ) + requests.append(request) + worse_prompts.append(worse_p) + better_prompts.append(better_p) + + responses = self._llm_query(requests) + + self._per_field_short_hints = [] + combined_strings = [] + for response in responses: + hints = self._parse_per_field_hints(response) + self._per_field_short_hints.append(hints) + combined_strings.append( + f"task_description: {hints['task']}\n" + f"system_behavior: {hints['role']}\n" + f"output_constraints: {hints['constraints']}" + ) + + return combined_strings, worse_prompts, better_prompts + + def _long_term_reflection(self, short_term_reflections: List[str]) -> None: + request = self._long_term_template.format( + PROBLEM_DESCRIPTION=self.problem_description, + TOP_PROMPTS_HISTORY=self._format_top_prompts_history(), + PRIOR_TASK_HINT=self._per_field_long_hints["task"], + PRIOR_ROLE_HINT=self._per_field_long_hints["role"], + PRIOR_CONSTRAINTS_HINT=self._per_field_long_hints["constraints"], + NEW_SHORT_TERM_REFLECTIONS="\n---\n".join(short_term_reflections), + ) + response = self._llm_query([request])[0] + hints = self._parse_per_field_hints(response) + self._per_field_long_hints = hints + self._long_term_reflection_str = ( + f"task_description: {hints['task']}\n" + f"system_behavior: {hints['role']}\n" + f"output_constraints: {hints['constraints']}" + ) + + def _crossover( + self, + short_term_reflection_tuple: Tuple[ + List[str], List[Prompt], List[Prompt] + ], + ) -> List[Prompt]: + _, worse_prompts, better_prompts = short_term_reflection_tuple + requests = [] + for i, (worse_p, better_p) in enumerate( + zip(worse_prompts, better_prompts) + ): + hints = ( + self._per_field_short_hints[i] + if i < len(self._per_field_short_hints) + else { + "task": self._FALLBACK_HINT, + "role": self._FALLBACK_HINT, + "constraints": self._FALLBACK_HINT, + } + ) + request = self._crossover_template.format( + PROBLEM_DESCRIPTION=self.problem_description, + WORSE_PROMPT_TEXT=worse_p.text, + WORSE_PROMPT_ROLE=worse_p.role, + WORSE_PROMPT_CONSTRAINTS=worse_p.constraints, + BETTER_PROMPT_TEXT=better_p.text, + BETTER_PROMPT_ROLE=better_p.role, + BETTER_PROMPT_CONSTRAINTS=better_p.constraints, + TASK_HINT=hints["task"], + ROLE_HINT=hints["role"], + CONSTRAINTS_HINT=hints["constraints"], + WORSE_SCORE=self._format_score(worse_p.score), + BETTER_SCORE=self._format_score(better_p.score), + ) + requests.append(request) + + responses = self._llm_query(requests) + crossed_population = [] + for i, response in enumerate(responses): + fields = self._parse_3f_response( + response, + fallback_text=better_prompts[i].text, + fallback_role=better_prompts[i].role, + fallback_constraints=better_prompts[i].constraints, + ) + crossed_population.append( + Prompt( + text=fields["task_description"], + role=fields["system_behavior"], + constraints=fields["output_constraints"], + origin=PromptOrigin.EVOLUTED, + ) + ) + + assert len(crossed_population) == self.population_size + return crossed_population + + def _mutate(self) -> List[Prompt]: + hints = self._per_field_long_hints + request = self._mutation_template.format( + PROBLEM_DESCRIPTION=self.problem_description, + TASK_HINT=hints["task"], + ROLE_HINT=hints["role"], + CONSTRAINTS_HINT=hints["constraints"], + ELITIST_PROMPT_TEXT=self.elitist.text, + ELITIST_PROMPT_ROLE=self.elitist.role, + ELITIST_PROMPT_CONSTRAINTS=self.elitist.constraints, + ELITIST_SCORE=self._format_score(self.elitist.score), + BAD_EXAMPLES=self._format_bad_examples(), + ) + responses = self._llm_query([request] * self.population_size) + mutated_population = [] + for response in responses: + fields = self._parse_3f_response( + response, + fallback_text=self.elitist.text, + fallback_role=self.elitist.role, + fallback_constraints=self.elitist.constraints, + ) + mutated_population.append( + Prompt( + text=fields["task_description"], + role=fields["system_behavior"], + constraints=fields["output_constraints"], + origin=PromptOrigin.MUTATED, + ) + ) + return mutated_population diff --git a/coolprompt/optimizer/reflective_prompt/factorized_evoluter.py b/coolprompt/optimizer/reflective_prompt/factorized_evoluter.py new file mode 100644 index 00000000..cc380f8d --- /dev/null +++ b/coolprompt/optimizer/reflective_prompt/factorized_evoluter.py @@ -0,0 +1,349 @@ +import os +from typing import List, Optional, Tuple + +from langchain_core.language_models.base import BaseLanguageModel +from langchain_core.messages.ai import AIMessage + +from coolprompt.evaluator import Evaluator +from coolprompt.optimizer.reflective_prompt.coevo_base_evoluter import ( + ReflectiveEvoluter, +) +from coolprompt.utils.logging_config import logger +from coolprompt.utils.parsing import extract_json +from coolprompt.utils.prompt_templates.reflective_templates_factorized import ( + DEDUP_ROLE_TEMPLATE, + DEDUP_CONSTRAINTS_TEMPLATE, +) + + +class FactorizedEvoluter: + + def __init__( + self, + model: BaseLanguageModel, + evaluator: Evaluator, + train_dataset: List[str], + train_targets: List[str], + validation_dataset: List[str], + validation_targets: List[str], + problem_description: str, + initial_prompt: Optional[str] = None, + initial_role: Optional[str] = None, + initial_constraints: Optional[str] = None, + population_size: int = 5, + phase_epochs: Tuple[int, int, int] = (4, 3, 3), + run_constraints_phase: bool = True, + output_path: str = "./factorized_outputs", + use_cache: bool = True, + use_enhancements: bool = True, + use_dedup: bool = True, + val_evaluator: Optional[Evaluator] = None, + ) -> None: + self.model = model + self.evaluator = evaluator + self.val_evaluator = val_evaluator or evaluator + self.train_dataset = train_dataset + self.train_targets = train_targets + self.validation_dataset = validation_dataset + self.validation_targets = validation_targets + self.problem_description = problem_description + self.initial_prompt = initial_prompt + self.initial_role = initial_role or "" + self.initial_constraints = initial_constraints or "" + self.population_size = population_size + self.phase_epochs = phase_epochs + self.run_constraints_phase = run_constraints_phase and bool( + initial_constraints + ) + self.output_path = output_path + self.use_cache = use_cache + self.use_enhancements = use_enhancements + self.use_dedup = use_dedup + + self.best_prompt_overall = None + self.best_role_overall = None + self.best_constraints_overall = None + self.best_score_overall = None + self.candidates: List[dict] = [] + + def _make_phase_evoluter( + self, + phase_name: str, + initial_prompt: Optional[str], + initial_role: Optional[str], + initial_constraints: Optional[str], + num_epochs: int, + evolve_role: bool, + evolve_constraints: bool, + freeze_text: bool, + ) -> ReflectiveEvoluter: + text_only = ( + not evolve_role and not freeze_text and not evolve_constraints + ) + return ReflectiveEvoluter( + model=self.model, + evaluator=self.evaluator, + train_dataset=self.train_dataset, + train_targets=self.train_targets, + validation_dataset=self.validation_dataset, + validation_targets=self.validation_targets, + problem_description=self.problem_description, + initial_prompt=initial_prompt, + initial_role=initial_role, + initial_constraints=initial_constraints, + evolve_role=evolve_role, + evolve_constraints=evolve_constraints, + freeze_text=freeze_text, + text_only=text_only, + population_size=self.population_size, + num_epochs=num_epochs, + use_cache=self.use_cache, + output_path=os.path.join(self.output_path, phase_name), + use_enhancements=self.use_enhancements, + ) + + def _eval_val(self, prompt: str, role: str, constraints: str) -> float: + result = self.val_evaluator.evaluate( + prompt=prompt, + dataset=self.validation_dataset, + targets=self.validation_targets, + system_role=role or None, + constraints=constraints or None, + ) + assert isinstance(result, float) + return result + + def _field_ablation( + self, + best_text: Optional[str], + best_role: Optional[str], + best_constraints: str, + score_text_role_constraints: Optional[float] = None, + ) -> Tuple[List[dict], dict]: + logger.info( + "[Field Ablation] Evaluating all field combinations on validation set..." + ) + assert best_text is not None and best_role is not None + score_a = self._eval_val(best_text, "", "") + logger.info(f" text_only: {score_a:.4f}") + score_b = self._eval_val(best_text, best_role, "") + logger.info(f" text_role: {score_b:.4f}") + candidates = [ + { + "combo": "text_only", + "prompt": best_text, + "role": "", + "constraints": "", + "val_score": score_a, + }, + { + "combo": "text_role", + "prompt": best_text, + "role": best_role, + "constraints": "", + "val_score": score_b, + }, + ] + if best_constraints: + if score_text_role_constraints is None: + score_text_role_constraints = self._eval_val( + best_text, best_role, best_constraints + ) + logger.info( + f" text_role_constraints: {score_text_role_constraints:.4f}" + ) + candidates.append( + { + "combo": "text_role_constraints", + "prompt": best_text, + "role": best_role, + "constraints": best_constraints, + "val_score": score_text_role_constraints, + } + ) + best_c = max(candidates, key=lambda c: c["val_score"]) + logger.info(f"Best combo: {best_c['combo']} (val={best_c['val_score']:.4f})") + return candidates, best_c + + def _llm_call(self, request: str) -> str: + responses = self.model.batch([request]) + r = responses[0] + return r.content if isinstance(r, AIMessage) else r + + def _dedup_role(self, task_text: str, role: str) -> str: + if not role or not self.use_dedup: + return role + try: + parsed = extract_json( + self._llm_call( + DEDUP_ROLE_TEMPLATE.format(TASK=task_text, ROLE=role) + ) + ) + if parsed and "system_behavior" in parsed: + cleaned = str(parsed["system_behavior"]).strip() + if cleaned != role: + logger.info( + f"[Dedup role] seed cleaned: '{role[:80]}' '{cleaned[:80]}'" + ) + return cleaned + except Exception as e: + logger.warning(f"Dedup role failed: {e}. Using original.") + return role + + def _dedup_constraints( + self, task_text: str, role: str, constraints: str + ) -> str: + if not constraints or not self.use_dedup: + return constraints + try: + parsed = extract_json( + self._llm_call( + DEDUP_CONSTRAINTS_TEMPLATE.format( + TASK=task_text, + ROLE=role or "(none)", + CONSTRAINTS=constraints, + ) + ) + ) + if parsed and "output_constraints" in parsed: + cleaned = str(parsed["output_constraints"]).strip() + if cleaned != constraints: + logger.info( + f"[Dedup constraints] seed cleaned: '{constraints[:80]}' '{cleaned[:80]}'" + ) + return cleaned + except Exception as e: + logger.warning(f"Dedup constraints failed: {e}. Using original.") + return constraints + + def evolution(self) -> str: + last_phase = 3 if self.run_constraints_phase else 2 + + logger.info( + f"Factorized evolution: {self.phase_epochs[0]} + {self.phase_epochs[1]}" + + ( + f" + {self.phase_epochs[2]} epochs" + if self.run_constraints_phase + else " epochs" + ) + + f" | phases: text -> role" + + (" -> constraints" if self.run_constraints_phase else "") + ) + + logger.info( + f"[Phase 1/{last_phase}] Optimizing task_description ({self.phase_epochs[0]} epochs)" + ) + p1 = self._make_phase_evoluter( + phase_name="phase1_text", + initial_prompt=self.initial_prompt, + initial_role=None, + initial_constraints=None, + num_epochs=self.phase_epochs[0], + evolve_role=False, + evolve_constraints=False, + freeze_text=False, + ) + p1.evolution(skip_validation=True) + best_text = p1.best_prompt_overall + assert best_text is not None + logger.info(f"Phase 1 best text score (train): {p1.best_score_overall:.4f}") + logger.info(f"Phase 1 best text: {best_text[:120]}") + + logger.info( + f"[Phase 2/{last_phase}] Optimizing system_behavior ({self.phase_epochs[1]} epochs)" + ) + initial_role_for_p2 = self._dedup_role( + best_text, self.initial_role or "" + ) + skip_p2_val = self.run_constraints_phase + p2 = self._make_phase_evoluter( + phase_name="phase2_role", + initial_prompt=best_text, + initial_role=initial_role_for_p2, + initial_constraints=None, + num_epochs=self.phase_epochs[1], + evolve_role=True, + evolve_constraints=False, + freeze_text=True, + ) + p2.evolution(skip_validation=skip_p2_val) + best_role = p2.best_role_overall + logger.info(f"Phase 2 best role score (train): {p2.best_score_overall:.4f}") + logger.info(f"Phase 2 best role: {(best_role or '')[:120]}") + + if not self.run_constraints_phase: + self.initial_prompt = p1.initial_prompt + self.initial_role = p2.initial_role or "" + self.initial_constraints = "" + candidates, best_c = self._field_ablation( + best_text=best_text, + best_role=p2.best_role_overall, + best_constraints="", + ) + self.candidates = candidates + self.best_prompt_overall = best_c["prompt"] + self.best_role_overall = best_c["role"] + self.best_constraints_overall = best_c["constraints"] + self.best_score_overall = best_c["val_score"] + return self.best_prompt_overall + + val_text_only = self._eval_val(best_text, "", "") + val_text_role = self._eval_val(best_text, best_role or "", "") + logger.info( + f"[Pre-Phase 3 check] text_only val: {val_text_only:.4f}, text_role val: {val_text_role:.4f}" + ) + if val_text_only >= val_text_role: + logger.info("Role does not improve on validation. Skipping Phase 3.") + self.initial_prompt = p1.initial_prompt + self.initial_role = p2.initial_role or "" + self.initial_constraints = "" + candidates, best_c = self._field_ablation( + best_text=best_text, + best_role=best_role or "", + best_constraints="", + ) + self.candidates = candidates + self.best_prompt_overall = best_c["prompt"] + self.best_role_overall = best_c["role"] + self.best_constraints_overall = best_c["constraints"] + self.best_score_overall = best_c["val_score"] + return self.best_prompt_overall + + logger.info( + f"[Phase 3/{last_phase}] Optimizing output_constraints ({self.phase_epochs[2]} epochs)" + ) + initial_constraints_for_p3 = self._dedup_constraints( + best_text, best_role or "", self.initial_constraints + ) + if not initial_constraints_for_p3 and self.initial_constraints: + logger.info("[Dedup] constraints redundant with task/role, using fallback seed") + initial_constraints_for_p3 = "Return only the final answer." + p3 = self._make_phase_evoluter( + phase_name="phase3_constraints", + initial_prompt=best_text, + initial_role=best_role or "", + initial_constraints=initial_constraints_for_p3, + num_epochs=self.phase_epochs[2], + evolve_role=False, + evolve_constraints=True, + freeze_text=False, + ) + p3.evolution(skip_validation=False) + logger.info(f"Phase 3 best constraints score: {p3.best_score_overall:.4f}") + + self.initial_prompt = p1.initial_prompt + self.initial_role = p2.initial_role or "" + self.initial_constraints = p3.initial_constraints or "" + + candidates, best_c = self._field_ablation( + best_text=best_text, + best_role=p2.best_role_overall, + best_constraints=p3.best_constraints_overall or "", + score_text_role_constraints=p3.best_score_overall, + ) + self.candidates = candidates + self.best_prompt_overall = best_c["prompt"] + self.best_role_overall = best_c["role"] + self.best_constraints_overall = best_c["constraints"] + self.best_score_overall = best_c["val_score"] + return self.best_prompt_overall diff --git a/coolprompt/optimizer/reflective_prompt/prompt.py b/coolprompt/optimizer/reflective_prompt/prompt.py index d2c231a3..63183c99 100644 --- a/coolprompt/optimizer/reflective_prompt/prompt.py +++ b/coolprompt/optimizer/reflective_prompt/prompt.py @@ -33,10 +33,10 @@ class BadExample: input (str): input of the example. output (str): model output for the example. correct (str): correct output of the example. - """ - - def __init__(self, input: str, output: str, correct: str): - self.input = input + """ + + def __init__(self, input: str, output: str, correct: str): + self.input = input self.output = output self.correct = correct @@ -69,15 +69,17 @@ def from_dict(cls: Type["BadExample"], data: dict) -> "BadExample": ) -class Prompt: - """Prompt candidate with origin, score, and optional failed examples.""" - - def __init__( +class Prompt: + """Prompt candidate with origin, score, and optional failed examples.""" + + def __init__( self, text: str, origin: PromptOrigin = PromptOrigin.EVOLUTED, score: float = None, bad_examples: List[BadExample] = [], + role: str = "", + constraints: str = "", ) -> None: """Prompt class. @@ -88,12 +90,16 @@ def __init__( score (float, optional): prompt evaluation score. Defaults to None. bad_examples (List[BadExample]): a list of bad examples for the prompt. + role (str): system behavior / role for the model (CoEvo). Defaults to "". + constraints (str): output format constraints (CoEvo). Defaults to "". """ self.text = text self.origin = origin self.score = score self.bad_examples = bad_examples + self.role = role + self.constraints = constraints def set_score(self, new_score: float) -> None: """Records new prompt evaluation score. @@ -127,6 +133,10 @@ def to_dict(self) -> dict: "text": self.text, "origin": self.origin.name, } + if self.role: + result["role"] = self.role + if self.constraints: + result["constraints"] = self.constraints if self.score is not None: result["score"] = self.score if len(self.bad_examples) > 0: @@ -159,6 +169,8 @@ def from_dict( BadExample.from_dict(bad_example_data) for bad_example_data in data.get("bad_examples", []) ], + role=data.get("role", ""), + constraints=data.get("constraints", ""), ) def __str__(self) -> str: diff --git a/coolprompt/optimizer/reflective_prompt/run.py b/coolprompt/optimizer/reflective_prompt/run.py index 33adac8b..f85aad30 100644 --- a/coolprompt/optimizer/reflective_prompt/run.py +++ b/coolprompt/optimizer/reflective_prompt/run.py @@ -1,4 +1,4 @@ -from typing import List, Tuple, override +from typing import List, Optional, Tuple, override from langchain_core.language_models import BaseLanguageModel @@ -9,6 +9,7 @@ BenchmarkContext, ) from coolprompt.optimizer.reflective_prompt.evoluter import ReflectiveEvoluter +from coolprompt.optimizer.reflective_prompt.coevo_evoluter import CoevoEvoluter from coolprompt.utils.deprecation import warn_deprecated from coolprompt.utils.logging_config import logger @@ -75,7 +76,7 @@ def reflectiveprompt( class ReflectiveMethod(AutoPromptingMethod): - """Reflective prompting method for auto‑prompting.""" + """Reflective prompting method for auto-prompting.""" def optimize( self, @@ -129,3 +130,146 @@ def is_data_driven(self) -> bool: @override def name(self) -> str: return "reflective" + + +def coevo( + model: BaseLanguageModel, + dataset_split: Tuple[List[str], List[str], List[str], List[str]], + evaluator: Evaluator, + problem_description: str, + initial_prompt: Optional[str] = None, + initial_role: Optional[str] = None, + initial_constraints: Optional[str] = None, + use_enhancements: bool = True, + use_bad_examples: Optional[bool] = None, + **kwargs, +) -> dict: + """Runs CoevoEvoluter optimization β€” co-evolves task description, system behavior and output constraints. + + Args: + model (BaseLanguageModel): a LLM to use. + dataset_split (Tuple[List[str], List[str], List[str], List[str]]): + train/valid split of dataset and corresponding targets. + evaluator (Evaluator): evaluator to compute metrics. + problem_description (str): short description of the task to optimize. + initial_prompt (str, optional): initial task description. Defaults to None. + initial_role (str, optional): initial system behavior. Defaults to None. + initial_constraints (str, optional): initial output constraints. Defaults to None. + use_enhancements (bool): whether to use enhanced co-evolution templates. Defaults to True. + use_bad_examples (bool, optional): whether to feed systematic error examples into mutation. If None, follows use_enhancements. + **kwargs: additional parameters (population_size, num_epochs, output_path, use_cache). + + Returns: + dict: best evolved prompt with keys: + - task_description (str): goes into the human message. + - system_behavior (str): goes into the system message. + - output_constraints (str): appended to the human message. + """ + train_dataset, validation_dataset, train_targets, validation_targets = ( + dataset_split + ) + args = { + "population_size": 10, + "num_epochs": 5, + "output_path": "./coevo_outputs", + "use_cache": True, + } + args.update(kwargs) + evoluter = CoevoEvoluter( + model=model, + evaluator=evaluator, + train_dataset=train_dataset, + train_targets=train_targets, + validation_dataset=validation_dataset, + validation_targets=validation_targets, + problem_description=problem_description, + initial_prompt=initial_prompt, + initial_role=initial_role, + initial_constraints=initial_constraints, + use_enhancements=use_enhancements, + use_bad_examples=use_bad_examples, + population_size=args["population_size"], + num_epochs=args["num_epochs"], + output_path=args["output_path"], + use_cache=args["use_cache"], + ) + logger.info("Starting CoEvo optimization...") + logger.debug(f"Start prompt:\n{initial_prompt}") + logger.debug(f"Problem description:\n{problem_description}") + evoluter.evolution() + logger.info("CoEvo optimization completed") + return { + "task_description": evoluter.best_prompt_overall or "", + "system_behavior": evoluter.best_role_overall or "", + "output_constraints": evoluter.best_constraints_overall or "", + } + + +class CoevoMethod(AutoPromptingMethod): + """Co-evolution method: structured prompt of three fields + (task_description, system_behavior, output_constraints). + + ``optimize`` returns the task_description as the main prompt and exposes + the evolved role and constraints via ``last_role`` / ``last_constraints``, + which PromptTuner surfaces as final_role / final_constraints. + """ + + last_role: str = "" + last_constraints: str = "" + + def optimize( + self, + model, + initial_prompt, + dataset_split, + evaluator, + problem_description, + **kwargs, + ): + """Run CoEvo through the shared method interface.""" + result = coevo( + model=model, + dataset_split=dataset_split, + evaluator=evaluator, + problem_description=problem_description, + initial_prompt=initial_prompt, + **kwargs, + ) + self.last_role = result["system_behavior"] + self.last_constraints = result["output_constraints"] + return result["task_description"] + + def run_configured_benchmark( + self, + ctx: BenchmarkContext, + start_prompt: str, + ) -> str: + """Run CoEvo from a benchmark context.""" + problem_description = ctx.config.get("problem_description") + if problem_description is None: + generator = SyntheticDataGenerator(ctx._system_model) + problem_description = generator._generate_problem_description( + prompt=start_prompt + ) + mc = ctx.config["method"] + return self.optimize( + ctx.model, + start_prompt, + dataset_split=ctx.dataset_split, + evaluator=ctx.evaluator, + problem_description=problem_description, + population_size=mc.get("population_size", 10), + num_epochs=mc.get("num_epochs", 5), + output_path=mc.get("output_path", "./coevo_outputs"), + use_cache=mc.get("use_cache", True), + use_enhancements=mc.get("use_enhancements", True), + use_bad_examples=mc.get("use_bad_examples", None), + ) + + def is_data_driven(self) -> bool: + return True + + @property + @override + def name(self) -> str: + return "coevo" diff --git a/coolprompt/utils/prompt_templates/reflective_templates_coevo_enhanced.py b/coolprompt/utils/prompt_templates/reflective_templates_coevo_enhanced.py new file mode 100644 index 00000000..9d7a3645 --- /dev/null +++ b/coolprompt/utils/prompt_templates/reflective_templates_coevo_enhanced.py @@ -0,0 +1,149 @@ +PARAPHRASING_TEMPLATE_COEVO_ENH = """Create {NUM_PROMPTS} diverse initial variants of the following three-field configuration. + +Task: {PROBLEM_DESCRIPTION} + +Seed configuration: +task_description: {PROMPT} +system_behavior: {ROLE} +output_constraints: {CONSTRAINTS} + +Rules for each variant: +- Vary at least two fields meaningfully from the seed. +- "task_description": change wording, directness, or how the output format is stated β€” preserve the task intent. +- "system_behavior": vary the reasoning strategy, cognitive angle, or focus area. Can start "You are [role]" only if immediately followed by a concrete behavioral instruction. 8–25 words. Must NOT restate task content. +- "output_constraints": vary format rules β€” length limits, structure, what to include or exclude. Must NOT include reasoning instructions or decision strategies (those belong in system_behavior). +- Each variant must differ meaningfully from the others. + +Output JSON only: +{{ + "prompts": [ + {{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}}, + {{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}}, + ... + ] +}} +Output JSON data only. +""" + +SHORT_TERM_REFLECTION_TEMPLATE_COEVO_ENH = """You are an expert in prompt optimization. Compare two three-field configurations and identify what makes the better one score higher. + +Task: {PROBLEM_DESCRIPTION} + +[Worse configuration] (score: {WORSE_SCORE}) +task_description: {WORSE_PROMPT_TEXT} +system_behavior: {WORSE_PROMPT_ROLE} +output_constraints: {WORSE_PROMPT_CONSTRAINTS} + +[Better configuration] (score: {BETTER_SCORE}) +task_description: {BETTER_PROMPT_TEXT} +system_behavior: {BETTER_PROMPT_ROLE} +output_constraints: {BETTER_PROMPT_CONSTRAINTS} + +Analyze each field separately: +- task_description: what difference in wording, directness, or format specification matters? +- system_behavior: what difference in reasoning strategy, focus area, or decision rule matters? +- output_constraints: what difference in format rule, length limit, or exclusion matters? + +Then write ONE combined actionable hint (under 30 words) identifying the most impactful change. +Wrap the hint with . +""" + +LONG_TERM_REFLECTION_TEMPLATE_COEVO_ENH = """You are an expert in prompt optimization. Synthesize patterns from the best-performing configurations found so far. + +Task: {PROBLEM_DESCRIPTION} + +Best configurations found so far (ranked by score, best first): +{TOP_PROMPTS_HISTORY} + +Prior accumulated insight: +{PRIOR_LONG_TERM_REFLECTION} + +New per-field observations from recent comparisons: +{NEW_SHORT_TERM_REFLECTIONS} + +Study the top configurations above. Identify what distinguishes the highest-scoring ones: +- task_description: what phrasing, directness, or output-format specification appears in the highest-scoring configs? +- system_behavior: what reasoning strategy, focus angle, or decision heuristic appears in the highest-scoring configs? +- output_constraints: what format rule β€” strictness of brevity, structure, or exclusions β€” correlates with higher scores? + +Write ONE updated actionable hint (under 50 words) covering the strongest pattern across all three fields. +Wrap the hint with . +""" + +CROSSOVER_TEMPLATE_COEVO_ENH = """You are an expert in prompt optimization. Design an improved three-field prompt configuration. + +Task: {PROBLEM_DESCRIPTION} + +[Worse configuration] (score: {WORSE_SCORE}) +task_description: {WORSE_PROMPT_TEXT} +system_behavior: {WORSE_PROMPT_ROLE} +output_constraints: {WORSE_PROMPT_CONSTRAINTS} + +[Better configuration] (score: {BETTER_SCORE}) +task_description: {BETTER_PROMPT_TEXT} +system_behavior: {BETTER_PROMPT_ROLE} +output_constraints: {BETTER_PROMPT_CONSTRAINTS} + +[Key insight from comparing these configurations] +{SHORT_TERM_REFLECTION} + +Combine the strongest element from each configuration. You may take any field unchanged from either configuration, or write a new version of a field guided by the insight above. +Goal: score above {BETTER_SCORE}. + +Field rules (strictly enforced): +- "task_description": WHAT to do and what output format is expected. 1–2 sentences. No reasoning instructions. +- "system_behavior": HOW to approach the task β€” reasoning strategy, what to prioritize, specific checks, or default decisions when input is ambiguous. Can start "You are [brief role]" ONLY if immediately followed by a concrete behavioral instruction. 1–2 sentences, 8–25 words. Must NOT repeat task_description content. +- "output_constraints": OUTPUT FORMAT rules only β€” length limits, structure, what to include or exclude in the response. Must NOT include reasoning instructions, decision strategies, or content already stated in the other two fields. 1–2 short rules. + +Output JSON only: +{{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}} +""" + +MUTATION_TEMPLATE_COEVO_ENH = """You are an expert in prompt optimization. Generate a targeted mutation of the current best configuration. + +Task: {PROBLEM_DESCRIPTION} + +[Accumulated insight on what works for this task] +{LONG_TERM_REFLECTION} + +[Current best configuration] (score: {ELITIST_SCORE}) +task_description: {ELITIST_PROMPT_TEXT} +system_behavior: {ELITIST_PROMPT_ROLE} +output_constraints: {ELITIST_PROMPT_CONSTRAINTS} + +[Cases where the current configuration most often fails] +Each line shows: input | wrong output the model gave | correct answer. +{BAD_EXAMPLES} + +Before writing, diagnose which field is responsible for these failures: +- task_description issue: does the instruction fail to convey the right output scope, format, or distinction between cases? +- system_behavior issue: does the reasoning strategy fail to handle the specific input patterns shown above, or is it biased toward certain classes? +- output_constraints issue: does the model produce extra text, wrong structure, or wrong format that hurts scoring? + +Mutate the field(s) most responsible for the failures. The other fields may stay the same or be improved moderately. +Goal: score above {ELITIST_SCORE}. + +Field rules (strictly enforced): +- "task_description": WHAT to do and what output format is expected. 1–2 sentences. +- "system_behavior": HOW to approach the task β€” reasoning strategy, what to check, default decisions for ambiguous cases. Can start "You are [brief role]" ONLY if immediately followed by a concrete behavioral instruction. 1–2 sentences, 8–25 words. Must NOT repeat task_description content. +- "output_constraints": OUTPUT FORMAT rules only β€” length, structure, what to include or exclude in the response. Must NOT include reasoning instructions, decision strategies, or content already stated in other fields. 1–2 short rules. + +Output JSON only: +{{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}} +""" + +PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_ENH = """Generate a concise initial three-field prompt configuration for the following task. + +Task: {PROBLEM_DESCRIPTION} + +Output a JSON object with exactly three fields: +- "task_description": What the model should do and how to return the answer (1–2 sentences). Be specific about the output format (e.g., a label, a number, a single sentence, the exact expected structure). +- "system_behavior": How the model should approach the task β€” one concrete reasoning principle or focus area (1 sentence, 8–20 words). Must be a behavioral instruction, not just a role title. + Good: "Check for ambiguous cases before deciding." / "Focus on the key entity before forming a response." + Bad: "You are an expert." β€” vague, no behavioral instruction. +- "output_constraints": Output format rules only β€” what to include or exclude in the response (1–2 short rules, under 15 words total). + Must NOT include reasoning instructions or decision strategies β€” those belong in system_behavior. + +Keep all fields brief and generic. These are starting points to be refined by the optimizer. +Output JSON only. +""" diff --git a/coolprompt/utils/prompt_templates/reflective_templates_coevo_per_field.py b/coolprompt/utils/prompt_templates/reflective_templates_coevo_per_field.py new file mode 100644 index 00000000..d3555b36 --- /dev/null +++ b/coolprompt/utils/prompt_templates/reflective_templates_coevo_per_field.py @@ -0,0 +1,114 @@ +from coolprompt.utils.prompt_templates.reflective_templates_coevo_enhanced import ( + PARAPHRASING_TEMPLATE_COEVO_ENH as PARAPHRASING_TEMPLATE_COEVO_PF, + PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_ENH as PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_PF, +) + +SHORT_TERM_REFLECTION_TEMPLATE_COEVO_PF = """You are an expert in prompt optimization. Compare two three-field configurations and identify what makes the better one score higher. + +Task: {PROBLEM_DESCRIPTION} + +[Worse configuration] (score: {WORSE_SCORE}) +task_description: {WORSE_PROMPT_TEXT} +system_behavior: {WORSE_PROMPT_ROLE} +output_constraints: {WORSE_PROMPT_CONSTRAINTS} + +[Better configuration] (score: {BETTER_SCORE}) +task_description: {BETTER_PROMPT_TEXT} +system_behavior: {BETTER_PROMPT_ROLE} +output_constraints: {BETTER_PROMPT_CONSTRAINTS} + +Analyze what makes the better configuration score higher. Then write three separate actionable hints, one per field (under 20 words each). +Wrap each hint in its own tags: +- task_description hint: wrap with +- system_behavior hint: wrap with +- output_constraints hint: wrap with +""" + +LONG_TERM_REFLECTION_TEMPLATE_COEVO_PF = """You are an expert in prompt optimization. Synthesize patterns from the best-performing configurations found so far. + +Task: {PROBLEM_DESCRIPTION} + +Best configurations found so far (ranked by score, best first): +{TOP_PROMPTS_HISTORY} + +Prior accumulated per-field insights: +task_description: {PRIOR_TASK_HINT} +system_behavior: {PRIOR_ROLE_HINT} +output_constraints: {PRIOR_CONSTRAINTS_HINT} + +New per-field observations from recent comparisons: +{NEW_SHORT_TERM_REFLECTIONS} + +Study the top configurations above and update each per-field insight. Write one updated actionable hint per field (under 30 words each) covering the strongest pattern. +Wrap each hint in its own tags: +- task_description hint: wrap with +- system_behavior hint: wrap with +- output_constraints hint: wrap with +""" + +CROSSOVER_TEMPLATE_COEVO_PF = """You are an expert in prompt optimization. Design an improved three-field prompt configuration. + +Task: {PROBLEM_DESCRIPTION} + +[Worse configuration] (score: {WORSE_SCORE}) +task_description: {WORSE_PROMPT_TEXT} +system_behavior: {WORSE_PROMPT_ROLE} +output_constraints: {WORSE_PROMPT_CONSTRAINTS} + +[Better configuration] (score: {BETTER_SCORE}) +task_description: {BETTER_PROMPT_TEXT} +system_behavior: {BETTER_PROMPT_ROLE} +output_constraints: {BETTER_PROMPT_CONSTRAINTS} + +[Per-field insights from comparing these configurations] +task_description: {TASK_HINT} +system_behavior: {ROLE_HINT} +output_constraints: {CONSTRAINTS_HINT} + +Combine the strongest element from each configuration guided by the field-specific insights above. +You may take any field unchanged from either configuration, or write a new version guided by its insight. +Goal: score above {BETTER_SCORE}. + +Field rules (strictly enforced): +- "task_description": WHAT to do and what output format is expected. 1–2 sentences. No reasoning instructions. +- "system_behavior": HOW to approach the task β€” reasoning strategy, what to prioritize, specific checks, or default decisions when input is ambiguous. Can start "You are [brief role]" ONLY if immediately followed by a concrete behavioral instruction. 1–2 sentences, 8–25 words. Must NOT repeat task_description content. +- "output_constraints": OUTPUT FORMAT rules only β€” length limits, structure, what to include or exclude in the response. Must NOT include reasoning instructions, decision strategies, or content already stated in the other two fields. 1–2 short rules. + +Output JSON only: +{{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}} +""" + +MUTATION_TEMPLATE_COEVO_PF = """You are an expert in prompt optimization. Generate a targeted mutation of the current best configuration. + +Task: {PROBLEM_DESCRIPTION} + +[Accumulated per-field insights on what works for this task] +task_description: {TASK_HINT} +system_behavior: {ROLE_HINT} +output_constraints: {CONSTRAINTS_HINT} + +[Current best configuration] (score: {ELITIST_SCORE}) +task_description: {ELITIST_PROMPT_TEXT} +system_behavior: {ELITIST_PROMPT_ROLE} +output_constraints: {ELITIST_PROMPT_CONSTRAINTS} + +[Cases where the current configuration most often fails] +Each line shows: input | wrong output the model gave | correct answer. +{BAD_EXAMPLES} + +Before writing, diagnose which field is responsible for these failures: +- task_description: does the instruction fail to convey the right output scope, format, or distinction between cases? +- system_behavior: does the reasoning strategy fail to handle the specific input patterns shown above, or is it biased toward certain classes? +- output_constraints: does the model produce extra text, wrong structure, or wrong format that hurts scoring? + +Mutate the field(s) most responsible for the failures, guided by the per-field insights above. +Goal: score above {ELITIST_SCORE}. + +Field rules (strictly enforced): +- "task_description": WHAT to do and what output format is expected. 1–2 sentences. +- "system_behavior": HOW to approach the task β€” reasoning strategy, what to check, default decisions for ambiguous cases. Can start "You are [brief role]" ONLY if immediately followed by a concrete behavioral instruction. 1–2 sentences, 8–25 words. Must NOT repeat task_description content. +- "output_constraints": OUTPUT FORMAT rules only β€” length, structure, what to include or exclude in the response. Must NOT include reasoning instructions, decision strategies, or content already stated in other fields. 1–2 short rules. + +Output JSON only: +{{"task_description": "...", "system_behavior": "...", "output_constraints": "..."}} +""" diff --git a/coolprompt/utils/prompt_templates/reflective_templates_coevolution.py b/coolprompt/utils/prompt_templates/reflective_templates_coevolution.py new file mode 100644 index 00000000..c9ba550b --- /dev/null +++ b/coolprompt/utils/prompt_templates/reflective_templates_coevolution.py @@ -0,0 +1,351 @@ +REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_BASE = """You are an expert in prompt optimization. Your task is to give hints to design better prompt configurations. + +Below are two prompt configurations for {PROBLEM_DESCRIPTION}. +Each configuration has two components: +- system_behavior: behavioral instructions defining HOW the AI should reason, verify, and process information (NOT a persona or job title) +- task_description: the specific task instruction defining WHAT the AI should do + +The second configuration performs better than the first one. +[Worse configuration] +System Behavior: {WORSE_PROMPT_ROLE} +Task Description: {WORSE_PROMPT_TEXT} +[Better configuration] +System Behavior: {BETTER_PROMPT_ROLE} +Task Description: {BETTER_PROMPT_TEXT} +Analyze differences in both system_behavior and task_description separately. +Consider WHY the better configuration works better. Focus on actionable changes. +Respond with one concise hint (less than 30 words) covering what to change in system_behavior and what to change in task_description. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_BASE = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks. +Your response outputs a JSON object with two fields: "system_behavior" and "task_description". + +Write a new prompt configuration for the task: {PROBLEM_DESCRIPTION}. + +[Worse configuration] +System Behavior: {WORSE_PROMPT_ROLE} +Task Description: {WORSE_PROMPT_TEXT} +[Better configuration] +System Behavior: {BETTER_PROMPT_ROLE} +Task Description: {BETTER_PROMPT_TEXT} +[Reflection] +{SHORT_TERM_REFLECTION} +[Improved configuration] +Combine the strongest aspects of both configurations according to the reflection. +You may take the system_behavior approach from one configuration and the task_description approach from the other. + +Rules: +- "system_behavior" MUST describe specific ACTIONS and REASONING STEPS β€” not identity or expertise. + Do NOT write: "You are an expert in X", "A specialist in Y", "Domain professional" + DO write: "Before answering, verify X. Check for Y. If Z, then..." +- "system_behavior" MUST be a complete instruction of at least 8 words. +- "task_description" MUST be a clear, actionable task instruction. +- system_behavior and task_description must complement each other without duplicating instructions. +Output JSON data only. +""" + +REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_BASE = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks. +Your response outputs a JSON object with two fields: "system_behavior" and "task_description". + +Write a mutated prompt configuration for {PROBLEM_DESCRIPTION}. +[Prior reflection] +{LONG_TERM_REFLECTION} +[Current elitist configuration] +System Behavior: {ELITIST_PROMPT_ROLE} +Task Description: {ELITIST_PROMPT_TEXT} +[Mutated configuration] +IMPORTANT for system_behavior: Aggressively reimagine it. Create a fundamentally different behavioral specification. +Do NOT describe identity (who you are). Describe BEHAVIOR (what to do, what to check, how to reason). +system_behavior MUST be a complete instruction of at least 8 words β€” NOT a persona or job title. +Bad examples: "Topic Analyst", "Data Specialist", "You are an expert in X" +Good examples: "Before responding, verify the key facts in the input. Check for consistency and relevance.", "Identify the core information first, then formulate a concise and accurate response." +The task_description may be changed moderately, applying the accumulated reflection. +system_behavior and task_description must not repeat the same instructions. +Output JSON data only. +""" + +REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO ="""You are an expert in prompt optimization. Your task is to give hints to design better prompt configurations. + +Below are two prompt configurations for {PROBLEM_DESCRIPTION}. +Each configuration has two components: +- task_description: WHAT the model should do and how to format the answer. Covers the task itself and output requirements. +- system_behavior: HOW the model should approach the task β€” reasoning strategy, key things to check, special considerations. Can start with a brief role ("You are X") only if immediately followed by a concrete behavioral instruction. + +The second configuration performs better than the first one. +[Worse configuration] (score: {WORSE_SCORE}) +System Behavior: {WORSE_PROMPT_ROLE} +Task Description: {WORSE_PROMPT_TEXT} +[Better configuration] (score: {BETTER_SCORE}) +System Behavior: {BETTER_PROMPT_ROLE} +Task Description: {BETTER_PROMPT_TEXT} +Analyze differences in both components separately. +Consider WHY the better configuration scored higher on this specific task. +Respond with one concise hint (less than 30 words) about what makes the better configuration work. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO = """You are an expert in prompt optimization. Your task is to give hints to design better prompt configurations. + +Task: {PROBLEM_DESCRIPTION} + +Best-performing configurations found so far (ranked by score, best first): +{TOP_PROMPTS_HISTORY} + +Study these top configurations: what patterns in system_behavior and task_description do the higher-scoring ones share? + +Prior accumulated insight: +{PRIOR_LONG_TERM_REFLECTION} + +New observations from recent comparisons: +{NEW_SHORT_TERM_REFLECTIONS} + +Write one updated actionable hint (less than 50 words) about what makes a configuration score highest on this specific task. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks. +Your response outputs a JSON object with two fields: "system_behavior" and "task_description". + +Write a new prompt configuration for the task: {PROBLEM_DESCRIPTION}. + +[Worse configuration] (score: {WORSE_SCORE}) +System Behavior: {WORSE_PROMPT_ROLE} +Task Description: {WORSE_PROMPT_TEXT} +[Better configuration] (score: {BETTER_SCORE}) +System Behavior: {BETTER_PROMPT_ROLE} +Task Description: {BETTER_PROMPT_TEXT} +[Reflection] +{SHORT_TERM_REFLECTION} +[Improved configuration] +Combine the strongest aspects of both configurations according to the reflection. +Your goal is to score HIGHER than {BETTER_SCORE}. + +Field rules: +- "task_description": WHAT to do and how to format the answer. Clear and actionable. 1-2 sentences. +- "system_behavior": HOW to approach the task β€” reasoning strategy, what to prioritize, specific checks. + You CAN start with "You are [brief role]" if you immediately follow it with a concrete behavioral instruction. + Example: "You are a careful analyst. Focus on the most relevant detail and verify it matches the expected format." + NOT enough: "You are an expert." β€” must say what to DO or CHECK. + 1-2 sentences, 8-25 words. +- The two fields must cover different aspects β€” task_description covers the task itself, system_behavior covers the approach. Do not repeat the same instruction in both. +Output JSON data only. +""" + +REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks. +Your response outputs a JSON object with two fields: "system_behavior" and "task_description". + +Task: {PROBLEM_DESCRIPTION} +[Accumulated insight on what works for this task] +{LONG_TERM_REFLECTION} + +[Current best configuration] (score: {ELITIST_SCORE}) +System Behavior: {ELITIST_PROMPT_ROLE} +Task Description: {ELITIST_PROMPT_TEXT} + +[Examples where the current configuration most often fails] +Each example shows the input, the wrong answer the model gave, and the correct answer. +{BAD_EXAMPLES} + +Analyze these failure cases before writing: +- Does the failure come from the task instruction (task_description) or from the behavioral approach (system_behavior)? +- What specific reasoning step or focus is missing that would fix these cases? +- What change to either field would prevent these specific errors? + +Write a mutated configuration that directly addresses the identified failure pattern and aims to score above {ELITIST_SCORE}. + +Field rules: +- "task_description": WHAT to do and how to format the answer. 1-2 sentences. +- "system_behavior": HOW to approach the task β€” reasoning strategy, what to check, special considerations. + You CAN start with "You are [brief role]" if you immediately follow it with a concrete behavioral instruction. + Example: "You are a careful analyst. Focus on the most relevant detail before committing to an answer." + NOT enough: "You are an expert." β€” must say what to DO or CHECK. + 1-2 sentences, 8-25 words. +- The two fields must cover different aspects. Do not repeat the same instruction in both. +Output JSON data only. +""" + +REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO = """Create diverse variations of the given prompt configuration. +System Behavior: {ROLE} +Task Description: {PROMPT} + +Create {NUM_PROMPTS} variations with maximum diversity. +Each variation must have a meaningfully DIFFERENT system_behavior that takes a unique behavioral approach. + +Field rules: +- "task_description": WHAT to do and how to format the answer. Vary wording while preserving task intent. +- "system_behavior": HOW to approach the task β€” reasoning strategy, what to prioritize, specific checks. 1-2 sentences. + You CAN start with "You are [brief role]" if you immediately follow it with a concrete behavioral instruction. + Examples: "You are a precise reader. Focus on the key entity before formulating a response.", + "Before responding, identify the main requirement and verify your answer matches it.", + "You are a methodical solver. Break the input into parts and handle each systematically." + NOT enough: "You are an expert." β€” must state what to DO or CHECK. +- The two fields must cover different aspects. Do not repeat the same instruction in both. + +Output them in JSON structure below: +{{ + "prompts": [ + {{"system_behavior": "behavior 1", "task_description": "task 1"}}, + {{"system_behavior": "behavior 2", "task_description": "task 2"}}, + ... + {{"system_behavior": "behavior {NUM_PROMPTS}", "task_description": "task {NUM_PROMPTS}"}}, + ] +}} +Output JSON data only. +""" + +REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO = """Generate a simple initial prompt configuration for the task: {PROBLEM_DESCRIPTION}. + +Output a JSON object with exactly two fields: +- "system_behavior": A SHORT behavioral hint (1 sentence, max 15 words) about HOW to approach the task. + Write a simple practical instruction, NOT a role or identity. + Bad: "You are an expert in X", "Data Analyst" + Good: "Think step by step before answering.", "Check your reasoning carefully." +- "task_description": A SHORT task instruction (1 sentence, max 15 words) about WHAT to do. + +IMPORTANT: Keep BOTH fields brief and generic. These are starting points that will be refined later. +Do NOT write elaborate multi-step strategies. Do NOT include specific examples or edge cases. +Output JSON only. +""" + +REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_COEVO_3F = """You are an expert in prompt optimization. Your task is to give hints to design better prompt configurations. + +Below are two prompt configurations for {PROBLEM_DESCRIPTION}. +Each configuration has three components: +- task_description: WHAT the model should do and how to format the answer. Covers the task itself and output requirements. +- system_behavior: HOW the model should approach the task β€” reasoning strategy, what to check, special considerations. Can start with a brief role ("You are X") only if immediately followed by a concrete behavioral instruction. +- output_constraints: FORMAT and STYLE rules only β€” length limits, structure, tone, what to omit. Examples: "One sentence only.", "No extra context.", "Use subject-verb-object order." + +The second configuration performs better than the first one. +[Worse configuration] (score: {WORSE_SCORE}) +System Behavior: {WORSE_PROMPT_ROLE} +Task Description: {WORSE_PROMPT_TEXT} +Output Constraints: {WORSE_PROMPT_CONSTRAINTS} +[Better configuration] (score: {BETTER_SCORE}) +System Behavior: {BETTER_PROMPT_ROLE} +Task Description: {BETTER_PROMPT_TEXT} +Output Constraints: {BETTER_PROMPT_CONSTRAINTS} +Analyze differences in all three components separately. +Consider WHY the better configuration scored higher on this specific task. +Respond with one concise hint (less than 30 words) about what makes the better configuration work. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_COEVO_3F = """You are an expert in prompt optimization. Your task is to give hints to design better prompt configurations. + +Task: {PROBLEM_DESCRIPTION} + +Best-performing configurations found so far (ranked by score, best first): +{TOP_PROMPTS_HISTORY} + +Study these top configurations: what patterns in system_behavior, task_description, and output_constraints do the higher-scoring ones share? + +Prior accumulated insight: +{PRIOR_LONG_TERM_REFLECTION} + +New observations from recent comparisons: +{NEW_SHORT_TERM_REFLECTIONS} + +Write one updated actionable hint (less than 50 words) about what makes a configuration score highest on this task. +Cover three aspects: what system_behavior patterns work best, what task_description patterns work best, and what output_constraints are most effective. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_COEVO_3F = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks. +Your response outputs a JSON object with three fields: "system_behavior", "task_description", and "output_constraints". + +Write a new prompt configuration for the task: {PROBLEM_DESCRIPTION}. + +[Worse configuration] (score: {WORSE_SCORE}) +System Behavior: {WORSE_PROMPT_ROLE} +Task Description: {WORSE_PROMPT_TEXT} +Output Constraints: {WORSE_PROMPT_CONSTRAINTS} +[Better configuration] (score: {BETTER_SCORE}) +System Behavior: {BETTER_PROMPT_ROLE} +Task Description: {BETTER_PROMPT_TEXT} +Output Constraints: {BETTER_PROMPT_CONSTRAINTS} +[Reflection] +{SHORT_TERM_REFLECTION} +[Improved configuration] +Combine the strongest aspects of both configurations according to the reflection. +Your goal is to score HIGHER than {BETTER_SCORE}. + +Field rules: +- "task_description": WHAT to do and how to format the answer. 1-2 sentences. Clear and actionable. +- "system_behavior": HOW to approach the task β€” reasoning strategy, what to prioritize, specific checks. + You CAN start with "You are [brief role]" if immediately followed by a concrete behavioral instruction. + Example: "You are a careful analyst. Focus on the most relevant detail and verify it matches the expected format." + NOT enough: "You are an expert." β€” must say what to DO or CHECK. 1-2 sentences, 8-25 words. +- "output_constraints": FORMAT and STYLE rules only β€” length, structure, what to omit. 1-2 short rules. + Example: "One sentence only. No extra context beyond the main point." +- The three fields must cover different aspects β€” no repeated instructions across fields. +Output JSON data only. +""" + +REFLECTIVEPROMPT_MUTATION_TEMPLATE_COEVO_3F = """You are an expert in prompt optimization. Your task is to design prompt configurations that effectively solve tasks. +Your response outputs a JSON object with three fields: "system_behavior", "task_description", and "output_constraints". + +Task: {PROBLEM_DESCRIPTION} +[Accumulated insight on what works for this task] +{LONG_TERM_REFLECTION} + +[Current best configuration] (score: {ELITIST_SCORE}) +System Behavior: {ELITIST_PROMPT_ROLE} +Task Description: {ELITIST_PROMPT_TEXT} +Output Constraints: {ELITIST_PROMPT_CONSTRAINTS} + +[Examples where the current configuration most often fails] +Each example shows the input, the wrong answer the model gave, and the correct answer. +{BAD_EXAMPLES} + +Analyze these failure cases before writing: +- Does the failure come from task_description (wrong instruction), system_behavior (wrong approach), or output_constraints (wrong format rule)? +- What specific change to which field would prevent these errors? + +Write a mutated configuration that directly addresses the identified failure pattern and aims to score above {ELITIST_SCORE}. + +Field rules: +- "task_description": WHAT to do and how to format the answer. 1-2 sentences. +- "system_behavior": HOW to approach the task β€” reasoning strategy, what to check, special considerations. + You CAN start with "You are [brief role]" if immediately followed by a concrete behavioral instruction. + Example: "You are a careful analyst. Focus on the most relevant detail before committing to an answer." + NOT enough: "You are an expert." β€” must say what to DO or CHECK. 1-2 sentences, 8-25 words. +- "output_constraints": FORMAT and STYLE rules only β€” length limits, structure, what to include or omit. + Example: "One sentence only. No additional context. Start directly with the answer." +- The three fields must cover different aspects. No repeated instructions across fields. +Output JSON data only. +""" + +REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_COEVO_3F = """Create diverse variations of the given prompt configuration. +System Behavior: {ROLE} +Task Description: {PROMPT} +Output Constraints: {CONSTRAINTS} +Create {NUM_PROMPTS} variations with maximum diversity. +Each variation must have meaningfully DIFFERENT system_behavior, task_description, and output_constraints. +- "system_behavior": Start with a role identity, then describe reasoning steps. +- "task_description": Vary wording while preserving task intent. +- "output_constraints": Vary format, length, tone, or quality rules. +Output them in JSON structure below: +{{ + "prompts": [ + {{"system_behavior": "New behavior 1", "task_description": "New task 1", "output_constraints": "New constraints 1"}}, + {{"system_behavior": "New behavior 2", "task_description": "New task 2", "output_constraints": "New constraints 2"}}, + ... + {{"system_behavior": "New behavior {NUM_PROMPTS}", "task_description": "New task {NUM_PROMPTS}", "output_constraints": "New constraints {NUM_PROMPTS}"}}, + ] +}} +Output JSON data only. +""" + +REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_COEVO_3F = """Generate a simple initial prompt configuration for the task: {PROBLEM_DESCRIPTION}. + +Output a JSON object with exactly three fields: +- "system_behavior": A brief role identity + practical reasoning hint (max 20 words). + Example: "You are a logical analyst. Think step by step before answering." +- "task_description": A SHORT task instruction (1 sentence, max 15 words). +- "output_constraints": Brief rules about output format or style (max 15 words). + Example: "Be concise. Follow the requested format strictly." + +IMPORTANT: Keep ALL fields brief and generic. These are starting points that will be refined later. +Output JSON only. +""" diff --git a/coolprompt/utils/prompt_templates/reflective_templates_factorized.py b/coolprompt/utils/prompt_templates/reflective_templates_factorized.py new file mode 100644 index 00000000..7423da1f --- /dev/null +++ b/coolprompt/utils/prompt_templates/reflective_templates_factorized.py @@ -0,0 +1,305 @@ + +REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_ROLE_ONLY = """Create {NUM_PROMPTS} diverse system_behavior variants for a fixed task instruction. + +Task: {PROBLEM_DESCRIPTION} +Task instruction (frozen β€” do not change): {PROMPT} + +Current system_behavior: +{ROLE} + +Rules: +- Each system_behavior must be 1 sentence, between 8 and 25 words. +- You CAN start with "You are X" if you immediately follow it with a concrete behavior or focus. + It is NOT enough to only name a role β€” you must say what the model should DO or NOTICE. +- Vary the framing: some variants should name a focus area, some a cognitive strategy, some a domain angle. +- The variants should differ meaningfully from each other. +- Do NOT copy examples from below β€” they are for illustration only, from unrelated domains. + Example (translation task): "You are a careful translator. Preserve the original register and avoid literal word-for-word rendering." + Example (legal task): "Identify the main obligation being described before formulating an answer." + Example (coding task): "You are a code reviewer. Flag potential edge cases as well as the obvious issue." + +Output JSON: +{{ + "prompts": [ + {{"system_behavior": "variant 1"}}, + {{"system_behavior": "variant 2"}}, + ... + {{"system_behavior": "variant {NUM_PROMPTS}"}} + ] +}} +Output JSON data only. +""" + +REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_ROLE_ONLY = """You are an expert in prompt optimization. Give a hint for writing better system_behavior instructions. + +Task: {PROBLEM_DESCRIPTION} +Task instruction (fixed): {FROZEN_PROMPT_TEXT} + +Two system_behavior instructions were tested. +[Worse system_behavior] (score: {WORSE_SCORE}) +{WORSE_PROMPT_ROLE} +[Better system_behavior] (score: {BETTER_SCORE}) +{BETTER_PROMPT_ROLE} + +Why does the better framing lead to higher scores on this specific task? +Respond with one hint in less than 20 words. Focus on what the better framing emphasizes. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_ROLE_ONLY = """You are an expert in prompt optimization. Synthesize hints for writing better system_behavior instructions. + +Task: {PROBLEM_DESCRIPTION} + +Best-performing system_behavior instructions found so far (ranked by score, best first): +{TOP_PROMPTS_HISTORY} + +Study these top system_behavior instructions: what framing, cognitive strategy, or focus angle do the higher-scoring ones use that lower-scoring ones lack? + +Prior accumulated insight: +{PRIOR_LONG_TERM_REFLECTION} + +New observations from recent comparisons: +{NEW_SHORT_TERM_REFLECTIONS} + +Write one updated actionable hint (less than 40 words) about what framing in system_behavior scores highest on this specific task. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_ROLE_ONLY = """You are an expert in prompt optimization. Design a better system_behavior instruction. + +Task: {PROBLEM_DESCRIPTION} +Task instruction (frozen): {FROZEN_PROMPT_TEXT} + +[Worse system_behavior] +{WORSE_PROMPT_ROLE} +[Better system_behavior] +{BETTER_PROMPT_ROLE} +[Reflection] +{SHORT_TERM_REFLECTION} + +Write a new system_behavior that takes the best aspect of both. +Rules: +- 1-2 sentences, 8-25 words total. +- You CAN start with "You are X" if you follow it with a concrete behavior. + Pure labels without behavior (e.g., "You are an expert.") score poorly. +- Do NOT use domain-specific terms from the examples below β€” they are from unrelated tasks: + "Identify the key entity being described before forming a response." + "Weigh the broader context before committing to a specific category." + "You are a precise reader. Prioritize the most prominent feature over peripheral details." +Output JSON: {{"system_behavior": ""}} +Output JSON data only. +""" + +REFLECTIVEPROMPT_MUTATION_TEMPLATE_ROLE_ONLY = """You are an expert in prompt optimization. Generate a new system_behavior instruction. + +Task: {PROBLEM_DESCRIPTION} +Task instruction (frozen): {FROZEN_PROMPT_TEXT} + +[Accumulated insight on what framing works for this task] +{LONG_TERM_REFLECTION} + +[Current best system_behavior] (score: {ELITIST_SCORE}) +{ELITIST_PROMPT_ROLE} + +[Examples where the current system_behavior most often fails] +Each example shows the input, the wrong answer the model gave, and the correct answer. +{BAD_EXAMPLES} + +Analyze these failure cases before writing: +- What type of inputs or edge cases does the current system_behavior fail to handle? +- Is there a pattern in the errors (e.g., the model misjudges a specific input type)? +- What cognitive strategy or focus angle could help the model handle these cases correctly? + +Write a new system_behavior that directly addresses the identified failure pattern. +Rules: +- 1-2 sentences, 8-25 words total. +- You CAN start with "You are X" if you follow it with a concrete behavior or focus angle. + Pure persona labels alone (e.g., "You are an expert.") are not useful. +- Short roles generalize better β€” avoid multi-clause chains. +- Do NOT use domain-specific terms from the examples below β€” they are from unrelated tasks: + "Identify the key entity being described before forming a response." + "Weigh the broader context before committing to a specific category." + "You are a precise reader. Prioritize the most prominent feature over peripheral details." +Output JSON: {{"system_behavior": ""}} +Output JSON data only. +""" + + +REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_CONSTRAINTS_ONLY = """Create diverse output_constraints variants for a fixed prompt configuration. + +Task: {PROBLEM_DESCRIPTION} +Task Description (frozen): {PROMPT} +System Behavior (frozen): {ROLE} + +Current output_constraints to vary from: +{CONSTRAINTS} + +Create {NUM_PROMPTS} output_constraints variants with maximum diversity. +output_constraints must only contain OUTPUT FORMAT rules: what the response must or must not contain, length limits, and structural requirements. +Do NOT include in output_constraints: +- Reasoning instructions ("think step by step", "analyze X before deciding") +- Classification strategies ("use X as the default", "prefer Y when ambiguous") β€” those belong in system_behavior. +- Instructions that repeat what the task_description or system_behavior already says. +Do NOT repeat instructions already present in the task_description or system_behavior. +Examples of good output_constraints (output format rules only): +- "Return only the final answer. Do not include any explanation or preamble." +- "Write no more than one sentence. Use plain language." +- "Output only the requested value with no surrounding text." + +Output JSON: +{{ + "prompts": [ + {{"output_constraints": "variant 1"}}, + {{"output_constraints": "variant 2"}}, + ... + {{"output_constraints": "variant {NUM_PROMPTS}"}} + ] +}} +Output JSON data only. +""" + +REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY = """You are an expert in prompt optimization. Your task is to give hints for designing better output constraints. + +Task: {PROBLEM_DESCRIPTION} +Task Description (fixed): {FROZEN_PROMPT_TEXT} +System Behavior (fixed): {FROZEN_PROMPT_ROLE} + +Two output_constraints configurations were tested. The second performs better. +[Worse output_constraints] (score: {WORSE_SCORE}) +{WORSE_PROMPT_CONSTRAINTS} +[Better output_constraints] (score: {BETTER_SCORE}) +{BETTER_PROMPT_CONSTRAINTS} + +Analyze WHY the better output FORMAT rule leads to higher scores on this task. +Consider: does stricter output brevity help? Does removing explanation noise improve parsing? Does a cleaner response structure match the evaluation metric better? +Note: output_constraints should cover FORMAT only (length, structure, what to include/exclude in the response). Do NOT suggest classification strategies or reasoning instructions β€” those belong in system_behavior. +Respond with one concise hint (less than 30 words). +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_CONSTRAINTS_ONLY = """You are an expert in prompt optimization. Your task is to give hints for designing better output constraints. + +Task: {PROBLEM_DESCRIPTION} +Task Description (fixed): {FROZEN_PROMPT_TEXT} +System Behavior (fixed): {FROZEN_PROMPT_ROLE} + +Best-performing output_constraints found so far (ranked by score, best first): +{TOP_PROMPTS_HISTORY} + +Study these top constraints: what format rules (brevity, structure, what to exclude) do the higher-scoring ones enforce that lower-scoring ones do not? + +Prior accumulated insight: +{PRIOR_LONG_TERM_REFLECTION} + +New observations from recent comparisons: +{NEW_SHORT_TERM_REFLECTIONS} + +Write one updated actionable hint (less than 50 words) summarizing what output constraint patterns score highest on this specific task. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_CONSTRAINTS_ONLY = """You are an expert in prompt optimization. Your task is to design better output constraints. + +Task: {PROBLEM_DESCRIPTION} +Task Description (frozen): {FROZEN_PROMPT_TEXT} +System Behavior (frozen): {FROZEN_PROMPT_ROLE} + +[Worse output_constraints] (score: {WORSE_SCORE}) +{WORSE_PROMPT_CONSTRAINTS} +[Better output_constraints] (score: {BETTER_SCORE}) +{BETTER_PROMPT_CONSTRAINTS} +[Reflection] +{SHORT_TERM_REFLECTION} + +Write new output_constraints that combine the strongest aspects of both, targeting a score above {BETTER_SCORE}. +Rules: +- Must only cover OUTPUT FORMAT: what the response must/must not contain, length, structure. +- Must NOT contain reasoning instructions, chain-of-thought requirements, or classification strategies (which label to choose when uncertain) β€” those belong in system_behavior. +- Must NOT repeat instructions already in task_description or system_behavior. +- Keep it concise (1-2 sentences). +Output JSON: {{"output_constraints": ""}} +Output JSON data only. +""" + +REFLECTIVEPROMPT_MUTATION_TEMPLATE_CONSTRAINTS_ONLY = """You are an expert in prompt optimization. Your task is to design new output constraints. + +Task: {PROBLEM_DESCRIPTION} +Task Description (frozen): {FROZEN_PROMPT_TEXT} +System Behavior (frozen): {FROZEN_PROMPT_ROLE} + +[Accumulated insight on what constraint patterns work best] +{LONG_TERM_REFLECTION} + +[Current best output_constraints] (score: {ELITIST_SCORE}) +{ELITIST_PROMPT_CONSTRAINTS} + +[Examples where the current constraints most often fail] +Each example shows the input, the wrong answer the model gave, and the correct answer. +{BAD_EXAMPLES} + +Analyze these failure cases before writing: +- Does the model output contain extra text, explanation, or preamble that hurts evaluation? +- Is the response format wrong (e.g., wrong delimiter, extra whitespace, wrong casing)? +- Would stricter length or structure rules prevent these specific errors? + +Generate output_constraints that directly address the identified format failures and target a score above {ELITIST_SCORE}. +Rules: +- Must only cover OUTPUT FORMAT: what the response must/must not contain, length, structure. +- Must NOT contain reasoning instructions, chain-of-thought requirements, or classification strategies β€” those belong in system_behavior. +- Must NOT repeat instructions already in task_description or system_behavior. +- Try varying: length limits ("only the digit", "one sentence max"), format rules ("no preamble", "no explanation"), structural requirements. +- Keep it concise (1-2 sentences). +Output JSON: {{"output_constraints": ""}} +Output JSON data only. +""" + + + +DEDUP_ROLE_TEMPLATE = """You are a prompt engineer reviewing a multi-field prompt for redundancy. + +The task_description field is fixed: +TASK_DESCRIPTION: {TASK} + +Review this system_behavior: +SYSTEM_BEHAVIOR: {ROLE} + +Remove from SYSTEM_BEHAVIOR any content already covered by TASK_DESCRIPTION. +Specifically remove: +- Any restatement of what the task is β€” already in task_description +- Any label or value definitions already listed in task_description +- Any output format rules (e.g. "return only X") already in task_description +Keep only what is UNIQUE to system_behavior: cognitive strategy, reasoning approach, focus angle, default decisions when uncertain, persona framing. +If nothing unique remains, return an empty string. + +Example of what to remove (translation task, for illustration only): + task_description: "Translate the text from English to French." + system_behavior: "You are a translator. Translate English text to French. Preserve tone." + β†’ Remove "Translate English text to French" (already in task). Keep "Preserve tone." + +Return JSON only: {{"system_behavior": ""}} +Output JSON data only.""" + +DEDUP_CONSTRAINTS_TEMPLATE = """You are a prompt engineer reviewing a multi-field prompt for redundancy. + +The task_description and system_behavior fields are fixed: +TASK_DESCRIPTION: {TASK} +SYSTEM_BEHAVIOR: {ROLE} + +Review these output_constraints: +OUTPUT_CONSTRAINTS: {CONSTRAINTS} + +Remove from OUTPUT_CONSTRAINTS any content already covered by TASK_DESCRIPTION or SYSTEM_BEHAVIOR. +Specifically remove: +- Any label or value definitions already in task_description +- Any output format rules already specified in task_description +- Any reasoning instructions or decision strategies already in system_behavior +Keep only UNIQUE format rules: response length limits, structural requirements, style restrictions not already stated elsewhere. +If nothing unique remains, return an empty string. + +Example of what to remove (legal task, for illustration only): + task_description: "Identify the main obligation. Return a single sentence." + output_constraints: "Return a single sentence stating the main obligation. Be concise." + β†’ Remove "Return a single sentence" (already in task). Keep "Be concise" only if it adds something new. + +Return JSON only: {{"output_constraints": ""}} +Output JSON data only.""" diff --git a/coolprompt/utils/prompt_templates/reflective_templates_fixed_role.py b/coolprompt/utils/prompt_templates/reflective_templates_fixed_role.py new file mode 100644 index 00000000..a297cb67 --- /dev/null +++ b/coolprompt/utils/prompt_templates/reflective_templates_fixed_role.py @@ -0,0 +1,89 @@ +REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_FIXED_ROLE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts. + +Below are two prompt configurations for {PROBLEM_DESCRIPTION}. +You are provided with two prompt versions below, where the second version performs better than the first one. + +The System Role is FIXED and is the same for both prompts: +Role: {BETTER_PROMPT_ROLE} + +[Worse prompt text] +Prompt: {WORSE_PROMPT_TEXT} +[Better prompt text] +Prompt: {BETTER_PROMPT_TEXT} + +You respond only with one small hint for designing better prompts TEXT, based on the two prompt versions and fixed role, using less than 20 words. +I want you to generate only one new hint for the prompt text itself. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_FIXED_ROLE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts. + +Below is your prior longβˆ’term reflection on designing prompts for {PROBLEM_DESCRIPTION}. +{PRIOR_LONG_TERM_REFLECTION} + +Below are some newly gained insights. +{NEW_SHORT_TERM_REFLECTIONS} + +Write the constructive hint for designing better prompt TEXTS, based on prior reflections and new insights and using less than 50 words. +The System Role is FIXED, so focus only on optimizing the user prompt text. +I want you to generate only one new constructive hint. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_FIXED_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems. +Your response outputs a JSON object with one field: "prompt". + +The System Role is FIXED and cannot be changed: +Role: {BETTER_PROMPT_ROLE} + +Write a new prompt text for the task: {PROBLEM_DESCRIPTION}. + +[Worse prompt text] +Prompt: {WORSE_PROMPT_TEXT} +[Better prompt text] +Prompt: {BETTER_PROMPT_TEXT} +[Reflection] +{SHORT_TERM_REFLECTION} +[Improved prompt configuration] +Please write an improved prompt text, according to the reflection, optimized for the fixed role above. +Output JSON data only. +""" + +REFLECTIVEPROMPT_MUTATION_TEMPLATE_FIXED_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems. +Your response outputs a JSON object with one field: "prompt". + +The System Role is FIXED and cannot be changed: +Role: {ELITIST_PROMPT_ROLE} + +Write a mutated prompt text for {PROBLEM_DESCRIPTION}. +[Prior reflection] +{LONG_TERM_REFLECTION} +[Current elitist prompt text] +Prompt: {ELITIST_PROMPT_TEXT} +[Mutated prompt configuration] +Please write a mutated prompt text, according to the reflection, optimized for the fixed role above. +Output JSON data only. +""" + +REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_FIXED_ROLE = """Paraphrase the given prompt text keeping its initial meaning. The role is fixed. + +Fixed Role: {ROLE} +Prompt: {PROMPT} + +Create {NUM_PROMPTS} new variations of this prompt text (optimized for the fixed role) and output them in JSON structure below: +{{ + "prompts": [ + "New prompt 1", + "New prompt 2", + ... + "New prompt {NUM_PROMPTS}" + ] +}} +Output JSON data only. +""" + +REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_FIXED_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems. +Write a prompt text that will effectively solve the task: {PROBLEM_DESCRIPTION}. +The System Role is FIXED and provided separately. Focus only on the prompt text. +Output a JSON object with one field: "prompt". +""" diff --git a/coolprompt/utils/prompt_templates/reflective_templates_no_role.py b/coolprompt/utils/prompt_templates/reflective_templates_no_role.py new file mode 100644 index 00000000..e8ba276c --- /dev/null +++ b/coolprompt/utils/prompt_templates/reflective_templates_no_role.py @@ -0,0 +1,76 @@ +REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_NO_ROLE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts. + +Below are two prompts for {PROBLEM_DESCRIPTION}. +You are provided with two prompt versions below, where the second version performs better than the first one. +[Worse prompt] +{WORSE_PROMPT_TEXT} +[Better prompt] +{BETTER_PROMPT_TEXT} +You respond only with one small hint for designing better prompts , based on the two prompt versions and using less than 20 words. +I want you to generate only one new hint. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_NO_ROLE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts. + +Below is your prior long-term reflection on designing prompts for {PROBLEM_DESCRIPTION}. +{PRIOR_LONG_TERM_REFLECTION} + +Below are some newly gained insights. +{NEW_SHORT_TERM_REFLECTIONS} + +Write the constructive hint for designing better prompts, based on prior reflections and new insights and using less than 50 words. +I want you to generate only one new constructive hint. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_NO_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems. +Your response outputs prompt text and nothing else. + +Write a prompt for the task: {PROBLEM_DESCRIPTION}. + +[Worse prompt] +{WORSE_PROMPT_TEXT} +[Better prompt] +{BETTER_PROMPT_TEXT} +[Reflection] +{SHORT_TERM_REFLECTION} +[Improved prompt] +Please write an improved prompt, according to the reflection. +Bracket the final prompt with . +""" + +REFLECTIVEPROMPT_MUTATION_TEMPLATE_NO_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems. +Your response outputs prompt text and nothing else. + +Write a prompt for {PROBLEM_DESCRIPTION}. +[Prior reflection] +{LONG_TERM_REFLECTION} +[Prompt] +{ELITIST_PROMPT_TEXT} +[Improved prompt] +Please write a mutated prompt, according to the reflection. +Output prompt only. +Bracket the final prompt with . +""" + +REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_NO_ROLE = """Paraphrase the given prompt text keeping its initial meaning. +Prompt: {PROMPT} +Create the new variations of this prompt and output them in JSON structure below: +{{ + "prompts": [ + "New prompt 1", + "New prompt 2", + "New prompt 3", + ... + "New prompt {NUM_PROMPTS}", + ] +}} +Output JSON data only. +""" + +REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_NO_ROLE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems. +Write a prompt that will effectively solve the task: {PROBLEM_DESCRIPTION}. +Output prompt only. +Bracket the final prompt with . +""" diff --git a/coolprompt/utils/prompt_templates/reflective_templates_orig.py b/coolprompt/utils/prompt_templates/reflective_templates_orig.py new file mode 100644 index 00000000..a6393b7b --- /dev/null +++ b/coolprompt/utils/prompt_templates/reflective_templates_orig.py @@ -0,0 +1,76 @@ +REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts. + +Below are two prompts for {PROBLEM_DESCRIPTION}. +You are provided with two prompt versions below, where the second version performs better than the first one. +[Worse prompt] +{WORSE_PROMPT} +[Better prompt] +{BETTER_PROMPT} +You respond only with one small hint for designing better prompts , based on the two prompt versions and using less than 20 words. +I want you to generate only one new hint. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE = """You are an expert in the domain of optimization prompts. Your task is to give hints to design better prompts. + +Below is your prior longβˆ’term reflection on designing prompts for {PROBLEM_DESCRIPTION}. +{PRIOR_LONG_TERM_REFLECTION} + +Below are some newly gained insights. +{NEW_SHORT_TERM_REFLECTIONS} + +Write the constructive hint for designing better prompts, based on prior reflections and new insights and using less than 50 words. +I want you to generate only one new constructive hint. For example, you can try to recommend word replacements, active/positive voice conversions, adding words or delete words. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_CROSSOVER_TEMPLATE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems. +Your response outputs prompt text and nothing else. + +Write a prompt for the task: {PROBLEM_DESCRIPTION}. + +[Worse prompt] +{WORSE_PROMPT} +[Better prompt] +{BETTER_PROMPT} +[Reflection] +{SHORT_TERM_REFLECTION} +[Improved prompt] +Please write an improved prompt, according to the reflection. +Bracket the final prompt with . +""" + +REFLECTIVEPROMPT_MUTATION_TEMPLATE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems. +Your response outputs prompt text and nothing else. + +Write a prompt for {PROBLEM_DESCRIPTION}. +[Prior reflection] +{LONG_TERM_REFLECTION} +[Prompt] +{ELITIST_PROMPT} +[Improved prompt] +Please write a mutated prompt, according to the reflection. +Output prompt only. +Bracket the final prompt with . +""" + +REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE = """Paraphrase the given prompt text keeping its initial meaning. +Prompt: {PROMPT} +Create the new variations of this prompt and output them in JSON structure below: +{{ + "prompts": [ + "New prompt 1", + "New prompt 2", + "New prompt 3", + ... + "New prompt {NUM_PROMPTS}", + ] +}} +Output JSON data only. +""" + +REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE = """You are an expert in the domain of optimization prompts. Your task is to design prompts that can effectively solve optimization problems. +Write a prompt that will effectively solve the task: {PROBLEM_DESCRIPTION}. +Output prompt only. +Bracket the final prompt with . +""" diff --git a/coolprompt/utils/prompt_templates/reflective_templates_text_only.py b/coolprompt/utils/prompt_templates/reflective_templates_text_only.py new file mode 100644 index 00000000..6b2bfa8a --- /dev/null +++ b/coolprompt/utils/prompt_templates/reflective_templates_text_only.py @@ -0,0 +1,113 @@ +REFLECTIVEPROMPT_PARAPHRASING_TEMPLATE_TEXT_ONLY = """Create {NUM_PROMPTS} concise variations of the task instruction below. + +Task instruction: {PROMPT} + +Rules: +- Each variation must be 1-2 sentences only. +- Preserve the output format requirement exactly (e.g. numeric label, single word, etc.). +- Change wording, emphasis, or directness β€” do not add extra sentences or explanations. +- Do not include any system role or behavior description in the instruction. + +Output JSON: +{{ + "prompts": [ + "Variation 1", + "Variation 2", + ... + "Variation {NUM_PROMPTS}" + ] +}} +Output JSON data only. +""" + +REFLECTIVEPROMPT_SHORT_TERM_REFLECTION_TEMPLATE_TEXT_ONLY = """You are an expert in prompt optimization. Give a brief hint for writing better task instructions. + +Task: {PROBLEM_DESCRIPTION} + +Two task instructions were tested. +[Worse instruction] (score: {WORSE_SCORE}) +{WORSE_PROMPT_TEXT} +[Better instruction] (score: {BETTER_SCORE}) +{BETTER_PROMPT_TEXT} + +Why does the better instruction lead to higher scores? Focus on wording, directness, or clarity differences. +Give one actionable hint in less than 20 words. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_LONG_TERM_REFLECTION_TEMPLATE_TEXT_ONLY = """You are an expert in prompt optimization. Synthesize hints for writing better task instructions. + +Task: {PROBLEM_DESCRIPTION} + +Best-performing task instructions found so far (ranked by score, best first): +{TOP_PROMPTS_HISTORY} + +Study these top instructions carefully: what phrasing, verb choice, or framing distinguishes the higher-scoring ones? What do they have in common that lower-scoring instructions lack? + +Prior accumulated insight: +{PRIOR_LONG_TERM_REFLECTION} + +New observations from recent comparisons: +{NEW_SHORT_TERM_REFLECTIONS} + +Write one updated actionable hint (less than 40 words) about what makes a task instruction score higher on this specific task. +Bracket the final hint with . +""" + +REFLECTIVEPROMPT_CROSSOVER_TEMPLATE_TEXT_ONLY = """You are an expert in prompt optimization. Combine two task instructions into a better one. + +Task: {PROBLEM_DESCRIPTION} + +[Worse instruction] +{WORSE_PROMPT_TEXT} +[Better instruction] +{BETTER_PROMPT_TEXT} +[Reflection] +{SHORT_TERM_REFLECTION} + +Write a new, improved task instruction. +Rules: +- 1-2 sentences only. No role or behavior description. +- Keep the output format requirement (how the answer must be returned) from the better instruction. +- No motivational or filler language. +Bracket the final instruction with . +""" + +REFLECTIVEPROMPT_MUTATION_TEMPLATE_TEXT_ONLY = """You are an expert in prompt optimization. Improve the task instruction below. + +Task: {PROBLEM_DESCRIPTION} + +[Accumulated insight on what works for this task] +{LONG_TERM_REFLECTION} + +[Current best task instruction] (score: {ELITIST_SCORE}) +{ELITIST_PROMPT_TEXT} + +[Examples where the current instruction most often fails] +Each example shows the input, the wrong answer the model gave, and the correct answer. +{BAD_EXAMPLES} + +Analyze these failure cases before writing: +- What type of inputs does the model get wrong? +- Is there a common pattern (e.g., ambiguous phrasing, specific input type, edge case)? +- What does the correct answer reveal about what the instruction fails to convey? + +Write a mutated instruction that directly addresses the identified failure pattern. +Rules: +- 1-2 sentences only. No role or behavior description. +- Keep the output format requirement intact (e.g. numeric label, exact phrasing for the answer format). +- Do not add motivational phrases, emotional language, or explanations targeted at the reader. +- Must be meaningfully different from the current instruction. +Bracket the final instruction with . +""" + +REFLECTIVEPROMPT_PROMPT_BY_DESCRIPTION_TEMPLATE_TEXT_ONLY = """Write a concise task instruction for the following task. + +Task: {PROBLEM_DESCRIPTION} + +Rules: +- 1-2 sentences only. No role or behavior description. +- Include how the answer should be returned (output format). +- Be direct and clear. +Bracket the final instruction with . +""" diff --git a/coolprompt/utils/var_validation.py b/coolprompt/utils/var_validation.py index f5195245..4d76f55b 100644 --- a/coolprompt/utils/var_validation.py +++ b/coolprompt/utils/var_validation.py @@ -8,7 +8,10 @@ from coolprompt.optimizer.hyper.meta_prompt import HyPERLightMethod from coolprompt.optimizer.hyper.hyper import HyPERMethod from coolprompt.optimizer.prompt_compressor import CompressorMethod -from coolprompt.optimizer.reflective_prompt import ReflectiveMethod +from coolprompt.optimizer.reflective_prompt import ( + ReflectiveMethod, + CoevoMethod, +) from coolprompt.optimizer.regps import ReGPSMethod from coolprompt.optimizer.rider import RIDERGenesisMethod from coolprompt.utils.enums import PD_Method, Task @@ -18,6 +21,7 @@ "hyper_light": HyPERLightMethod, "hyper": HyPERMethod, "reflective": ReflectiveMethod, + "coevo": CoevoMethod, "distill": DistillMethod, "regps": ReGPSMethod, "compress": CompressorMethod, diff --git a/notebooks/examples/benchmark_coevo.png b/notebooks/examples/benchmark_coevo.png new file mode 100644 index 00000000..a479816d Binary files /dev/null and b/notebooks/examples/benchmark_coevo.png differ diff --git a/notebooks/examples/coevo_demo.ipynb b/notebooks/examples/coevo_demo.ipynb new file mode 100644 index 00000000..b6bce120 --- /dev/null +++ b/notebooks/examples/coevo_demo.ipynb @@ -0,0 +1,1058 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "b2e9afbd", + "metadata": {}, + "source": [ + "# CoEvo \u2014 \u043a\u0430\u043a \u043c\u0435\u0442\u043e\u0434 \u0443\u043b\u0443\u0447\u0448\u0430\u0435\u0442 \u043f\u0440\u043e\u043c\u043f\u0442\n", + "\n", + "CoEvo \u0430\u0432\u0442\u043e\u043c\u0430\u0442\u0438\u0447\u0435\u0441\u043a\u0438 \u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0438\u0440\u0443\u0435\u0442 \u043f\u0440\u043e\u043c\u043f\u0442: \u0440\u0430\u0441\u043a\u043b\u0430\u0434\u044b\u0432\u0430\u0435\u0442 \u0435\u0433\u043e \u043d\u0430 \u0442\u0440\u0438 \u043f\u043e\u043b\u044f \u2014 **\u0440\u043e\u043b\u044c**, **\u043e\u043f\u0438\u0441\u0430\u043d\u0438\u0435\n", + "\u0437\u0430\u0434\u0430\u0447\u0438** \u0438 **\u043e\u0433\u0440\u0430\u043d\u0438\u0447\u0435\u043d\u0438\u044f \u043d\u0430 \u043e\u0442\u0432\u0435\u0442** \u2014 \u0438 \u044d\u0432\u043e\u043b\u044e\u0446\u0438\u043e\u043d\u0438\u0440\u0443\u0435\u0442 \u0438\u0445 \u0432\u043c\u0435\u0441\u0442\u0435, \u0438\u0441\u043f\u043e\u043b\u044c\u0437\u0443\u044f \u0440\u0435\u0444\u043b\u0435\u043a\u0441\u0438\u044e (\u043c\u043e\u0434\u0435\u043b\u044c\n", + "\u0441\u0430\u043c\u0430 \u0430\u043d\u0430\u043b\u0438\u0437\u0438\u0440\u0443\u0435\u0442, \u0447\u0435\u043c \u0443\u0434\u0430\u0447\u043d\u044b\u0435 \u043f\u0440\u043e\u043c\u043f\u0442\u044b \u043e\u0442\u043b\u0438\u0447\u0430\u044e\u0442\u0441\u044f \u043e\u0442 \u043d\u0435\u0443\u0434\u0430\u0447\u043d\u044b\u0445).\n", + "\n", + "\u0417\u0434\u0435\u0441\u044c \u0431\u0435\u0440\u0451\u043c \u043f\u0440\u043e\u0441\u0442\u043e\u0439 \u043f\u0440\u043e\u043c\u043f\u0442 \u0434\u043b\u044f \u0437\u0430\u0434\u0430\u0447\u0438 \u00ab\u043e\u0442\u0432\u0435\u0442\u044c \u043d\u0430 \u0432\u043e\u043f\u0440\u043e\u0441 \u043f\u043e \u043a\u043e\u043d\u0442\u0435\u043a\u0441\u0442\u0443\u00bb \u0438 \u0441\u043c\u043e\u0442\u0440\u0438\u043c, \u043d\u0430\u0441\u043a\u043e\u043b\u044c\u043a\u043e\n", + "\u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0430\u0446\u0438\u044f \u043f\u043e\u0434\u043d\u0438\u043c\u0430\u0435\u0442 \u043a\u0430\u0447\u0435\u0441\u0442\u0432\u043e (BERTScore).\n", + "\n", + "> \u0414\u043b\u044f \u0437\u0430\u043f\u0443\u0441\u043a\u0430 \u043d\u0443\u0436\u0435\u043d `OPENAI_API_KEY` \u0438 `pip install coolprompt datasets`." + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "id": "063da7ad", + "metadata": { + "execution": { + "iopub.execute_input": "2026-07-24T09:54:13.265675Z", + "iopub.status.busy": "2026-07-24T09:54:13.265509Z", + "iopub.status.idle": "2026-07-24T09:54:20.287984Z", + "shell.execute_reply": "2026-07-24T09:54:20.287556Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\u0433\u043e\u0442\u043e\u0432\u043e\n" + ] + } + ], + "source": [ + "import os\n", + "from pathlib import Path\n", + "# \u043f\u043e\u0434\u0445\u0432\u0430\u0442\u044b\u0432\u0430\u0435\u043c \u043b\u043e\u043a\u0430\u043b\u044c\u043d\u044b\u0439 .env, \u0435\u0441\u043b\u0438 \u043e\u043d \u0435\u0441\u0442\u044c (\u0438\u043d\u0430\u0447\u0435 \u0434\u043e\u0441\u0442\u0430\u0442\u043e\u0447\u043d\u043e \u043f\u0435\u0440\u0435\u043c\u0435\u043d\u043d\u043e\u0439 \u043e\u043a\u0440\u0443\u0436\u0435\u043d\u0438\u044f)\n", + "for _p in [Path.cwd(), *Path.cwd().parents]:\n", + " _env = _p / \".env\"\n", + " if _env.exists():\n", + " for _line in _env.read_text().splitlines():\n", + " if \"=\" in _line and not _line.lstrip().startswith(\"#\"):\n", + " _k, _v = _line.split(\"=\", 1); os.environ.setdefault(_k.strip(), _v.strip())\n", + " break\n", + "assert os.environ.get(\"OPENAI_API_KEY\"), \"\u041d\u0443\u0436\u0435\u043d OPENAI_API_KEY (\u0438\u043b\u0438 \u043b\u043e\u043a\u0430\u043b\u044c\u043d\u044b\u0439 .env)\"\n", + "\n", + "import logging, warnings\n", + "import matplotlib.pyplot as plt\n", + "from datasets import load_dataset\n", + "from langchain_openai import ChatOpenAI\n", + "from coolprompt.assistant import PromptTuner\n", + "from coolprompt.utils.logging_config import logger\n", + "for _h in logger.handlers:\n", + " _h.setLevel(logging.ERROR)\n", + "logger.propagate = False\n", + "warnings.filterwarnings(\"ignore\")\n", + "print(\"\u0433\u043e\u0442\u043e\u0432\u043e\")" + ] + }, + { + "cell_type": "markdown", + "id": "84062783", + "metadata": {}, + "source": [ + "## \u0414\u0430\u043d\u043d\u044b\u0435\n", + "\n", + "\u041f\u0430\u0440\u044b \u00ab\u0432\u043e\u043f\u0440\u043e\u0441 + \u043a\u043e\u043d\u0442\u0435\u043a\u0441\u0442 \u2192 \u043e\u0442\u0432\u0435\u0442\u00bb \u0438\u0437 \u043e\u0442\u043a\u0440\u044b\u0442\u043e\u0433\u043e QA-\u043d\u0430\u0431\u043e\u0440\u0430. \u041e\u0442\u0432\u0435\u0442\u044b \u043a\u043e\u0440\u043e\u0442\u043a\u0438\u0435, \u043f\u043e\u044d\u0442\u043e\u043c\u0443 \u043f\u0440\u043e\u043c\u043f\u0442,\n", + "\u043a\u043e\u0442\u043e\u0440\u044b\u0439 \u044d\u0442\u043e\u0433\u043e \u043d\u0435 \u0443\u0447\u0438\u0442\u044b\u0432\u0430\u0435\u0442, \u043b\u0435\u0433\u043a\u043e \u0442\u0435\u0440\u044f\u0435\u0442 \u0432 \u043a\u0430\u0447\u0435\u0441\u0442\u0432\u0435 \u2014 \u0435\u0441\u0442\u044c \u043a\u0443\u0434\u0430 \u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0438\u0440\u043e\u0432\u0430\u0442\u044c." + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "id": "d8ac94a9", + "metadata": { + "execution": { + "iopub.execute_input": "2026-07-24T09:54:20.289557Z", + "iopub.status.busy": "2026-07-24T09:54:20.289337Z", + "iopub.status.idle": "2026-07-24T09:54:23.020457Z", + "shell.execute_reply": "2026-07-24T09:54:23.019440Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\u043f\u0440\u0438\u043c\u0435\u0440\u043e\u0432: 60\n", + "\n", + "\u0432\u0445\u043e\u0434 : Context: The Normans (Norman: Nourmands; French: Normands; Latin: Normanni) were the people who in the 10th and 11th centuries gave their name to Norm ...\n", + "\u043e\u0442\u0432\u0435\u0442: France\n" + ] + } + ], + "source": [ + "ds = load_dataset(\"rajpurkar/squad_v2\", split=\"validation\").filter(\n", + " lambda r: len(r[\"answers\"][\"text\"]) > 0)\n", + "rows = ds.select(list(range(0, 5000, 80))[:60])\n", + "dataset = [f\"Context: {r['context']}\\nQuestion: {r['question']}\" for r in rows]\n", + "target = [r[\"answers\"][\"text\"][0] for r in rows]\n", + "\n", + "print(\"\u043f\u0440\u0438\u043c\u0435\u0440\u043e\u0432:\", len(dataset))\n", + "print(\"\\n\u0432\u0445\u043e\u0434 :\", dataset[0][:150], \"...\")\n", + "print(\"\u043e\u0442\u0432\u0435\u0442:\", target[0])" + ] + }, + { + "cell_type": "markdown", + "id": "11d82975", + "metadata": {}, + "source": [ + "## \u0417\u0430\u043f\u0443\u0441\u043a CoEvo\n", + "\n", + "\u0421\u0442\u0430\u0440\u0442\u0443\u0435\u043c \u0441 \u043e\u0431\u044b\u0447\u043d\u043e\u0433\u043e \u043f\u0440\u043e\u0441\u0442\u043e\u0433\u043e \u043f\u0440\u043e\u043c\u043f\u0442\u0430. `PromptTuner.run` \u0441\u0430\u043c \u0437\u0430\u043c\u0435\u0440\u0438\u0442 \u0435\u0433\u043e \u043a\u0430\u0447\u0435\u0441\u0442\u0432\u043e (baseline),\n", + "\u043f\u0440\u043e\u0432\u0435\u0434\u0451\u0442 \u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0430\u0446\u0438\u044e \u0438 \u0437\u0430\u043c\u0435\u0440\u0438\u0442 \u0438\u0442\u043e\u0433. \u0426\u0435\u043b\u0435\u0432\u0430\u044f \u043c\u043e\u0434\u0435\u043b\u044c \u2014 `gpt-4.1-nano`, \u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0430\u0442\u043e\u0440 \u2014 `gpt-4o-mini`.\n", + "\u041f\u043e\u043f\u0443\u0442\u043d\u043e \u0441\u043e\u0431\u0438\u0440\u0430\u0435\u043c \u043b\u0443\u0447\u0448\u0438\u0439 \u0440\u0435\u0437\u0443\u043b\u044c\u0442\u0430\u0442 \u043f\u043e \u044d\u043f\u043e\u0445\u0430\u043c \u0434\u043b\u044f \u0433\u0440\u0430\u0444\u0438\u043a\u0430." + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "id": "b2b4a75f", + "metadata": { + "execution": { + "iopub.execute_input": "2026-07-24T09:54:23.021695Z", + "iopub.status.busy": "2026-07-24T09:54:23.021600Z", + "iopub.status.idle": "2026-07-24T10:05:44.980594Z", + "shell.execute_reply": "2026-07-24T10:05:44.979980Z" + } + }, + "outputs": [ + { + "output_type": "stream", + "name": "stdout", + "text": [ + "\u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0430\u0446\u0438\u044f \u0437\u0430\u0432\u0435\u0440\u0448\u0435\u043d\u0430\n" + ] + } + ], + "source": [ + "START_PROMPT = \"Answer the question based on the context.\"\n", + "\n", + "target_model = ChatOpenAI(model=\"gpt-4.1-nano\", temperature=0, max_tokens=96, timeout=60, max_retries=3)\n", + "optimizer_model = ChatOpenAI(model=\"gpt-4o-mini\", temperature=0.7, max_tokens=1500, timeout=90, max_retries=3)\n", + "\n", + "tuner = PromptTuner(target_model=target_model, system_model=optimizer_model)\n", + "tuner.run(\n", + " start_prompt=START_PROMPT,\n", + " task=\"generation\",\n", + " metric=\"bertscore\",\n", + " dataset=dataset,\n", + " target=target,\n", + " method=\"coevo\",\n", + " system_model_as_optimizer=True,\n", + " validation_size=0.34,\n", + " population_size=4,\n", + " num_epochs=5,\n", + " verbose=0,\n", + ")\n", + "print(\"\u043e\u043f\u0442\u0438\u043c\u0438\u0437\u0430\u0446\u0438\u044f \u0437\u0430\u0432\u0435\u0440\u0448\u0435\u043d\u0430\")" + ] + }, + { + "cell_type": "markdown", + "id": "7099f268", + "metadata": {}, + "source": [ + "## \u0420\u0435\u0437\u0443\u043b\u044c\u0442\u0430\u0442: \u0431\u044b\u043b\u043e / \u0441\u0442\u0430\u043b\u043e" + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "id": "23f75979", + "metadata": { + "execution": { + "iopub.execute_input": "2026-07-24T10:05:44.986456Z", + "iopub.status.busy": "2026-07-24T10:05:44.986350Z", + "iopub.status.idle": "2026-07-24T10:05:44.990288Z", + "shell.execute_reply": "2026-07-24T10:05:44.989876Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\u0441\u0442\u0430\u0440\u0442\u043e\u0432\u044b\u0439 \u043f\u0440\u043e\u043c\u043f\u0442 : BERTScore = 0.8233\n", + "\u043f\u043e\u0441\u043b\u0435 CoEvo : BERTScore = 0.8963\n", + "\u043f\u0440\u0438\u0440\u043e\u0441\u0442 : +0.0730\n" + ] + } + ], + "source": [ + "init_score = tuner.init_metric\n", + "final_score = tuner.final_metric\n", + "print(f\"\u0441\u0442\u0430\u0440\u0442\u043e\u0432\u044b\u0439 \u043f\u0440\u043e\u043c\u043f\u0442 : BERTScore = {init_score:.4f}\")\n", + "print(f\"\u043f\u043e\u0441\u043b\u0435 CoEvo : BERTScore = {final_score:.4f}\")\n", + "print(f\"\u043f\u0440\u0438\u0440\u043e\u0441\u0442 : {final_score-init_score:+.4f}\")" + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "d753197f", + "metadata": { + "execution": { + "iopub.execute_input": "2026-07-24T10:05:44.991427Z", + "iopub.status.busy": "2026-07-24T10:05:44.991333Z", + "iopub.status.idle": "2026-07-24T10:05:44.994396Z", + "shell.execute_reply": "2026-07-24T10:05:44.994093Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\u0411\u044b\u043b\u043e \u2014 \u043e\u0434\u0438\u043d \u043f\u0440\u043e\u043c\u043f\u0442:\n", + " Answer the question based on the context. \n", + "\n", + "\u0421\u0442\u0430\u043b\u043e \u2014 \u0442\u0440\u0438 \u043f\u043e\u043b\u044f, \u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u043f\u043e\u0434\u043e\u0431\u0440\u0430\u043b CoEvo:\n", + " \u0440\u043e\u043b\u044c : You are a focused extractor; ensure the answer is directly from the context.\n", + " \u0437\u0430\u0434\u0430\u0447\u0430 : Extract the precise answer from the context provided. Respond in JSON format.\n", + " \u043e\u0433\u0440\u0430\u043d\u0438\u0447\u0435\u043d\u0438\u044f: Response must be in JSON format, limited to 100 characters, with only the answer as the value.\n" + ] + } + ], + "source": [ + "print(\"\u0411\u044b\u043b\u043e \u2014 \u043e\u0434\u0438\u043d \u043f\u0440\u043e\u043c\u043f\u0442:\")\n", + "print(\" \", tuner.init_prompt, \"\\n\")\n", + "print(\"\u0421\u0442\u0430\u043b\u043e \u2014 \u0442\u0440\u0438 \u043f\u043e\u043b\u044f, \u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u043f\u043e\u0434\u043e\u0431\u0440\u0430\u043b CoEvo:\")\n", + "print(\" \u0440\u043e\u043b\u044c :\", tuner.final_role)\n", + "print(\" \u0437\u0430\u0434\u0430\u0447\u0430 :\", tuner.final_prompt)\n", + "print(\" \u043e\u0433\u0440\u0430\u043d\u0438\u0447\u0435\u043d\u0438\u044f:\", tuner.final_constraints)" + ] + }, + { + "cell_type": "markdown", + "id": "9ead0ba1", + "metadata": {}, + "source": [ + "## \u0412\u0438\u0437\u0443\u0430\u043b\u0438\u0437\u0430\u0446\u0438\u044f" + ] + }, + { + "cell_type": "code", + "execution_count": 6, + "id": "f3273c64", + "metadata": { + "execution": { + "iopub.execute_input": "2026-07-24T10:05:44.995486Z", + "iopub.status.busy": "2026-07-24T10:05:44.995430Z", + "iopub.status.idle": "2026-07-24T10:05:45.228724Z", + "shell.execute_reply": "2026-07-24T10:05:45.228222Z" + } + }, + "outputs": [ + { + "output_type": "display_data", + "data": { + "image/png": "iVBORw0KGgoAAAANSUhEUgAAA38AAAJYCAYAAADSaV+1AAAAOnRFWHRTb2Z0d2FyZQBNYXRwbG90bGliIHZlcnNpb24zLjExLjEsIGh0dHBzOi8vbWF0cGxvdGxpYi5vcmcvctoD+AAAAAlwSFlzAAAViAAAFYgBxNdAoAAAXzhJREFUeJzt3QeUU1X39/FN770MvfciXXoTFAtFBUVQAQUExUZReFQUFfRRARVUFEQQpFhABJSqgNSHooJIUar03oZe8q593pX8b27KJDOZEu73s1ZWJjc3NyfJTDK/nHP2SeVyuVwCAAAAALippU7uBgAAAAAAEh/hDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAB0ib3A0AcPPZt2+fHD9+XC5cuCA5cuSQ0qVLS6ZMmRLlvtasWSOXLl0Kad9s2bJJrVq1xAmuXLkix44dM6cjR45Izpw5pW7dusndrBTp8uXLsnr1avNzhQoVpECBAsndJEfbtGmTnDx50vysr4W+JoHcuHFDdu7cKSdOnJA0adJI7ty5JW/evOZ9J1wXL1407116rAwZMki+fPmkaNGikhz+/PNP0w6rYsWKSalSpcJ6P0yVKpWkTZvWPB59DyhYsKBkyZJFUpKjR4/K/v37zeeFtk0fY7DX7+rVq7Jy5UrP5erVq5vHBiBELgCIgNWrV7s6derkypcvn0vfWqyn1KlTu2699VbXe++95zp16lREn+/ixYv73F+gU61atVw3o7Nnz7pGjRrl6tixo6tGjRquvHnz+jz2qlWrJnczU6zvv//e8zz9/vvvyd0cRzt48KArc+bMntfj66+/9rvfv//+63rsscdc2bJl8/u3XqBAAdc999zj+uijj4Le37Vr11xTpkxxNW7c2JU+fXqf48TExLi6devm2rx5syupXL161ZU/f36ftjRo0CAi74dFixY1z93atWtdyeHixYuuMWPGuB588EG/nxepUqUy79VTp04NeIzatWt79n/yySeTtP1AtCP8AUiQ2NhY16OPPhpyABs9enREn3Gnh799+/a5ypYtG9Lj37RpU3I3N0Xq2rWreX5KlCiR3E1xvCeeeMLz+1q6dGkTzuz++OMPV65cuRL8N79//35X3bp1QzqOfoH1+uuvu27cuJHor9GcOXMCtuOff/6J2PuhnvQLu3PnzrmS0u7du0NuX/fu3f0e49tvv/XskzZtWtfff/+dpI8BiGYM+wSQoOFyd955p6xYscLnupIlS5ohRqdOnZLt27eb4VlJQYcMBRqqVb58ebmZ6Bd4HTt2lH/++cezTYe9FS9eXPLnzy8xMTFSqFAhKVKkiJQoUcKcw9v169flxx9/ND+3a9eOpycZ7dq1S7744gvP5eeff94M5bTr1q2beV+x/93r77sOldS/B/3bCEaHlTZp0sTcp5UON9RhprGxsbJlyxbPcfT967XXXjNDQ99++21JTJMmTQp43VdffSVDhgwJ+/1Q2/3vv//K4cOHva6fNm2a/P333/Lrr79K5syZJTno8Fpt54EDB8zwT6vx48dLy5Yt5aGHHvLafv/995vPmN27d8u1a9fMczJlypQkbjkQnQh/AOLtxRdf9Al+LVq0kNGjR0vFihU923T+37hx4+Sdd94Jejz9R0s/zHWeWurUqc18n3Dn3Dz55JMyYMCAOPfTf3gOHjzouVy2bFkpXLiwz36HDh0y4dVNA62/EBmftuv9azvc9J9X6/MWl++++05WrVplfq5du7Z8+umnJuDpP3n6D1GePHnMfEt//0C7g8/y5cs9l7Xd+g9xMHpc62uux27cuLHffXU/3T8uOiepUaNGifp7EYi2UX8/1b333hvv4+jviZ708WrQ1vBtpXMK9cuScPiby6T/IOvvjc7p1Dms+g+wnocipb8eY8eO9bRPf68efPBBn322bt0qf/zxh+eyPj8LFy6UOnXqeLbpHFcNSSNHjgx4X7179/YKfjo3bujQofLCCy9IunTpzDadS/jwww/L//73P89+//3vf80XXk2bNvVs06C4fv16z2Wd32yfX7tjxw6vYKN/l/6eszNnzsicOXM8l/X5tX5xNnny5LDCn/39cOPGjTJo0CCZP3++Z9uGDRukT58+MmHChJCOGan3Tg11L730knku9XEqfd26du3q9ZgnTpzoE/50f/3iS18P93vhBx98YIIkgDgkd9cjgOi0Z88eV7p06byG6Nx2221mvkqw4T5LlizxO8/n6aefduXJk8dn2E+RIkVcr732WsChSfZhTjqvMBRfffWVz/Anf3S+j3W/8ePHR6ztOu/Fuv/DDz/sCkfbtm09t33qqadc5cqV82lD9uzZzWPYuXOnz+21XdZ9M2TIEOd96pxN622yZMkScN8cOXKENLRL9/MnIc9tqPr27WuOlzt3br9DDON6LgYPHuwqVqyYT/tKlizpevPNN12XLl0y+xYuXDis4Xh6cv+tfPPNN+Z3w99cTp0fVbNmTfP7HJeU/HpcuXLFzK9zH69ly5YhDYl86KGHAh5T2zJ27Fif7Vu2bDHPm/U4AwYMCPga6/xB6772tuk8Uev1OlzV7rnnngvpfUrba92vTZs2rjJlynhtW7FiRcDHHMr7oQ5d7dChg8+w1r/++suVFO+dR44ccU2fPj3g8Tt37uzzt+SP/Xl/9913Q2o/4HSEPwDxMmLECJ9/QkP958Hqf//7n99J//ZThQoVzPy2uP7Z6d27t/mn2d/p8OHDnttduHDB659hDTHnz5/3Orb+427fx/rPbULbntDwp/9whxoktDDGggULoib8JfS5DVWpUqXMcbp06RLW7bQAiL/QZz+525aQ8BfqbQMFmGh4PRYtWuR1HH1/8Ud/h637VatWLegXTv4MGzbM6xj6Jdbx48dD3j9NmjSmyFKgEKJhLb7hr1GjRl77TZo0yfWf//zHa1uvXr0CtjXUL8M0yNsL3OgXGaGIxHtnMG+99VbIxaoKFix4U8/pBhID6/wBiBfrcEFVqVIlcwqHzttp27atGT7mpsO99Dj2kubbtm2T++67L865gzr0sXnz5n5PixYt8hqaZR1KdP78eZk9e7bXsX766SczDMvtgQcekKxZsyZa28NlvW97SfiqVaua4Xtu586dkw4dOsjevXslqdgfb4MGDcwQr/r16we9XVI9t1pO3z30L5whn/pc3nXXXWZ4rVX27NnNUiL+htrqY9bH7j7Z/1Z0iK71ej3Zh3zq76w+fj2Wvr7p06f3un7EiBHmMUXj66FzzqxuvfVWv/vp8+selukexqjPxVtvvSVLliwJ+DdhXw7BqkaNGub5D0SHstuHS69bty7g/jqEND50KK11CQN9ffV51/cdq2+++SbsIcR2OgRTX/9g7+mBJPS9My6bN2/2umxvp5V1uK8OBz579mxI9wE4WqJESgA3PXuVvPbt24d9jCFDhngdQ4ferVu3zqsEvw5Hsu4zY8aMeFe3mzx5stdt16xZ43W9DqO0euCBB7yu//XXXyPa9h9++MHVtGlTz2no0KFhPX/WYXLuk3WYmw5vs/caWcuiJ3bPn15n3ffkyZNm+6FDh4L2NEXiuQ3FG2+8YW6bKVMmn56LYOw9QdrrPXz4cK9ho+7lN44dOxZntUI9tWvXLuD9ac/P0qVLfSpNnj592gxBtB5Hh15G4+uhQ8atz2ew12PgwIFB/851mKD+nm/YsMHv7XXZGev++ncejI4YsN/HtGnTAvb8ae9nfHr+tJqodZ/WrVt7rtOhpNbrvvvuuwQPg+/Zs6fXvuXLl3eFKiHvncHo75VW77T2yur7WFx/w+7T/PnzQ34MgFPR8wcgXrR6nJUuIhwu+7fFzz33nClc4qa9MVrVzcpaDMEf7Ymw96C4T1pQxUqLMlSuXNlzWYsguKsIahGHuXPneq4rU6aMV2GTSLRdv9VfunSp5/Tyyy9LuN/eW2n7evbs6bmsxWP+85//BG2DlfbWuNuiPRBa7VALi8SXLsZsZe2xSerfC39mzZplzm+//fawKh1+//33Xpe7d+8u/fv39+rt0yIszzzzjFlwPKG0V0t/f7UYihYO0QIky5Ytk99//92r50PpaxaNr4e1R1orbgZ7PbTaphZnCbSP9qCNGTPG9BJqW+2VP+3vXfYeVDt/723a2xWIu3hJuLTYiZW1x8/e+xesImiotAcv2PMSTELeOwPZtGmT3HPPPV5FiT788MOgRbDs74FJObIBiFaEPwDxYh8mdfTo0bCPof/IWtWrV89nH/s2+238VbezBirrSf/Jt3vsscc8P2vQmTlzpicYWP8Zsu6XWG0Pl30YVSht0IqDly5dChgO3ENktdqj/nOnQw8feeQRU2UyHBok7cEx1C8IkuK51efht99+i1eVT2uFVtWmTRtJTFrR9e677zaBUisr6mNv1qyZeZ3syw5Yh9pF0+thHa6py5UEo8Mq9YsSfQ21inDnzp1NFUl/oWvUqFFmH6tcuXJ5XT59+nScy0LY2Yfk2tsXLq0Ga12yxT3kM1D4mzdvnqdKbXxpVVSrYENf/Ynve6c/Gh71Pcf6OTJs2DDzfh6M/XcllGG/gNMR/gDEi/3bWJ0DE+48FHsI8VeyXudRWV24cEEi6dFHH/WaG6frXlnPlf5T2aVLlxTXdnvPRyhtCPcbft1X18/SJSCC9XbY2ddh07aF2tOUFM+tu9dPX9vWrVuHdVt7++IKKwmxYMEC0+un/+y7/750HceGDRv6nTuo89Gi8fWw3kfGjBlDuo2GuB49epjfT51rqI9Re8/svUH2HjUN0FbB5kkq7WG1C9Yb5W++o84TDcbek6evq85hc39xpXPZrL9n+kXN119/LfGlvaHuZWKs95kU753+5mnr36D7OdLbaGjXZSDC7b2M9HsscDMi/AGIF13ryt7joAvyxsX64Wxfk8nfkB37Nvv6aQmlx9OhRm5aNOKvv/7yKg5zxx13+CyQnhLabh9SGEobtEchUK+F9li4h8jqsD5rONDCKO5v9kOxb98+r8v2f8iDSYrn9ocffjDnGqLCXRvMvr/+viSWV1991WsYnP6jvGfPHrNmn4aCvn373hSvhzXYBOuJ0y8jAq1VqAFU1+WzrydqXzjcXsBFC/fYg5CVrq1npe8F1vBn73H0F0CsvXp22mumRVysNPjZC1bZeyATMvRTg6P9d8L+np5Y753WAKrrKmrvnvtLC/1Ca8aMGWbIdCjsz0kkhlkDNzvCH4B40WFo9m+/ddH3X375xe/++q3u888/77WQsL3KoH2RYf2nSL/Vt4qrMmF8WIcl6bf2OszROj/K37ClSLRdF0q2DkvVBazDYa+IqOHMPuzviy++8JmrE2hYmgZDd1u0J9fefuuCzaEs2G1VpUqVkG+b2L8XGi50zlx8F3a3L4CuC037qzKoISAhcybd86CsWrVq5fXPs/15iMbXwx4W/Q2ztP4O6nBkDS+BQqC9l1PnEFrpa64L01s9/fTTfnvn9H7cXxS4PfXUU15/Q/ZeT/27th5LexY1rAei8+OCPeZA1q5d6zMEORRa1dM+nFIrBD/44INJ8t7pDvFafXj48OFeX0ho1ddw/ibtz1ukv2ADbkrJXXEGQPTSCoT2hd616l/Hjh3N+lQLFy50ff31165nn33Ws0D16NGjPbfX6+1V9O69916zqLUuJFy/fn2v63RdKl1cPr7r/OnJ30Leuk6Yv8qZ7sqG7oW6rSLR9oSu86dVGjNnzux1jCpVqrgmTJjgmjlzpuvRRx8NWvHUXu1TX0v38zR37lyzwLT1el1/K65qn7/88ovP7fSkvw/WdgerLhmJ5zbURap37tzpCpdWLrS3T9dc1MqK+rxNmTLFVKTU3/lAa96FWu1Tf/+s+zVv3txU19S/K3ulTz21aNEi6l4PpessWo+xa9cuv/tZK2vq4utPPPGE+X3XKo/6vOjznjFjxoAVbt308furEjpy5EhzLK2m2bVrV58qplrJ016J9Pr162YdTet+zZo1MxVPP/roI7/vLdYqnPraW6/T9zRrFWDryb7W4ssvvxzS++G8efNcn332mbkv+2PSdQvjWyUzPu+dZ86c8am4mj17dvOaBHrftle6ddPHZz1OfNaaBZyG8AcgQbTkuf2frWAna/hTjz32WMi31X/M7MJZ6kFPgRYa7t+/v9/9n3766YCPPaFtT2j4U59++mnIbdDS8fqPaqDwF9dp/fr1cYa/Vq1a+dxOQ4r1n7e4wkYknttg3GXogy0eHZd+/fqF1LaEhj8NIMGOr8tzBAt/0fB6qHHjxnkdQ58ff+zLKsR10mU8/vnnH7/HevXVV8M6VtGiRV07duzwe6wePXoEva39PdId/nRxefsXaLqQfSDvv/++1776/md9LcN9P9Tnx74ETrjCfe/8888/w2qjnjRk+qMLu7v30S9bAoVEAP+H8AcgwTZu3Oi3F8J+ypMnj88/NvqhPmjQIJ9/gKwn/Vb9k08+8XvfkQp/+o2xv/1/++23gI87oW2PRPhTEydOdOXMmTNgG/Sbfu0huXDhgtftQg1/entdg0yFG/46depk1qOzCiVsJPS5DUR7Ity9NK+88oorvvSfzBEjRvisnWfvzTh69GiCwp/evnLlyn6Pr+FV1xIMJ/yltNfDTXsKdX0/97H69u3rd7+9e/e6qlWrFtLvbbFixczohGC0t7BcuXJBj6Pt0uct0JqNSq/TdfL83V6DoQYhf+FPewat27U309/oBOtrpT111tssW7Ys7PdDfUx33XWXee9OqHDfOyMV/nQtTevvY+fOnRP8WAAn+L8yTQAQT7fccouZ5K9z1hYvXmxK6GsZcq3gp/NttLqervN02223+VTy02pxWq5e59x89913sn79elOuW4so6LwcnTukZc4DFSnREvMlSpQIua3Wtdjsle66detm1ghzK1y4sNSoUSPgsRLa9kKFCpniKqFUEAyma9eu0r59e1MoQdfn0zlHOg9LS7dr+3UNNl1ry99zYb1/+2PTdletWtU8BnclQN1uvY292p4WitGiC1p6X+/XuhaYdW6h9Rj2JSsi8dwGonNS3fOx4jPfz03nfPXr18+s8adzLbVUvz7vOu9Jq3Hq76W2L9BadFpExfocBJqDp/vp/MuJEyeatuucTt2mf0tabVHv13qcatWqRdXr4abPmRZi0fcPpb/LI0aM8JmfqnPTtBiKFivRuWs6J1IL4Og8Tp0DqJVIda1Pfb+566674lzOQn8HdEkFnQOqc111Dp3OGdT7cC+FoM+HzmcOVkxEr9PnRJeV0HlrOqdN29GpUyfTlo8++sjrOS5atKinkJJ1uy4bEug9Sulz3bt3b9m8ebNnmz4HWo3X3/uhPn/6GunzoNVRdV5dhQoVzHNdsmRJiYRw3zuzZMkS8H0nEH/zlHX9Sev8Qv1bBBC3VJoAQ9gPAICo16tXL1P8RP/51iqPSDm04mXHjh09l/WLjAYNGiRLWzQEamEf97pxMTExpj2lS5dOlvbAl4b2OXPmmJ/1ddGKqvFZYxFwGqp9AgAcQ3v9tNdBe7CQsmjPZLly5TyXP/vss2Rri7bjp59+8vSCai+gLltw+PDhZGsT/o/2/OrC8G4DBw4k+AEhoucPAACkCN9++61nyQFdZ3LHjh1mqGdy0eHsw4YN8xpW+8EHHxA0kpmub6mvg9Ihzbq+YLDhsgD+D+EPAACkGDqX0b0Auc5nDbRWHJxJ5/ndd999Ehsbay7rfExddxZAaAh/AAAAAOAAzPkDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMJfhPXs2dOcAAAAACAlSZvcDbjZbN68ObmbAAAAAAA+6PkDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwgLTJ3QAAAAAgLrt27ZIFCxbIvn375Pr161K4cGFp0aKFVK5cOSJP3sWLF+WXX36RP//8U06cOCE3btyQnDlzSsWKFeX222+XHDlyBL39lStX5Oeff5bffvtNTp48KVmyZJFy5crJnXfeKXnz5g25Hf/++68sXrxY9u7dKxcuXJCCBQtKqVKlTBv0mEBCEP4AAACQYmkQe/rpp2X69Ol+r7/jjjtk7NixUrx48Xjfx4cffiivvvqqnD171u/1GTJkkH79+smbb74padKk8bl+ypQpMmDAADl8+HDYt7WGW93vhx9+8Ht95syZpW/fvjJ06NCwHhtglcrlcrm8tiBB6tevb85Xr17NMwkAAJAAZ86ckSZNmsimTZuC7le0aFFZsWKFFCtWLOz7eP/9903oCsVzzz0nH3zwgde2kSNHSv/+/eO87YMPPihff/213+vWr18vd911lxw/fjzoMbSnU3sFgfgi/EUY4Q8AACAyevXqZXr13HT4pG7LmDGjjB8/Xvbs2eO5rlWrVjJ//vywjq/DRwsUKOAVunQ46eOPPy7p06eXr776SrZv3+65Lm3atHLkyBHJnTu3ufzPP/+YYadXr171CmgtW7aUAwcOyOeffy6XLl3yXKeXu3fv7tUGHSJ6yy23mP3dsmfPbsJihQoVTADesWOHGZJapUoVwh8ShPAXYYQ/AACAhDt06JDpybt27ZpXD1mtWrXMz/v37zdz6nSuntu6deukdu3aId+Hzh+09xZu27ZNypcvb34+duyYGU5qvY/ly5dLo0aNzM+DBw/2GobZtGlTWbp0qddw0EceecRzuXTp0iYwpkqVyrNNew2199BNg6DObdRQag+q2gNao0aNkB8fYEe1TwAAAKQ4M2bM8Ap+GnrcwU8VKVLE9PZZBZoXGIjOo7PKlCmTJ/ipfPnymZ7AQLexD0e1t0eHclrt3LnTFISxFonRHkwr7W20Bz+l8wUJfkgowh8AAABSnLVr13pdrl69us8+9m3a8xeOPHnySKVKlTyXtYdv1apVnsvaS6dVN61hUKt/ul2+fNnreNbhn+5wZ2dt44YNG8ywTuvj0d7BadOmyaBBg+SFF16QUaNGydatW8N6XEAgVPsEAABAiqPVL6389YbFxMR4Xd69e3fY9zN69Ghp3bq1Z2inztdr06aNmfM3d+5cT6DT+X4axLR30E2DmtXMmTNl4MCBpsKnmjp1qt+hpoF6DnV+oC7roPMK7Tp27CifffZZnEtOAMHQ8wcAAIAU59y5c16XtciLnTWIKWsvWqhuu+02+d///mcCoNIQ+M0335jhl6dPnzbbGjduLMuWLZOHHnrI67adOnXyurxx40ZTAOaxxx4zQ0D9VQG1LiehxV7s8w39BT+llUK1jTr3D4gvwh8AAABSHGtRFOVvdTL7tmDr6AWiC6mPGzfOhLtAdKim9rrZw5oWftHqo/Z5fRMnTpSFCxf6PZZ1zqB9mKhKly6dPPHEE/LOO+9I27Ztva7T5SwmT54c8mMD7Bj2CQAAgBQnZ86cXpfPnz/vs499W7hDIjU8ag+dhiq3W2+91fSwaZD8+eefzRILOhxz0qRJpofwjz/+8OqF/OSTT6RkyZLy9ttv+/Q8VqtWTXLlyuVVAVSXq7Au6WD33nvvmfUErcM9tSfSbdasWdKtW7ewHifgRs8fAAAAUpyyZct6Xbaugxdomy79EA6d02cNflpNVAu+6BIOL730kgl/d955p+d6XfNPe/WsUqdObeb5HTx40KzBp2v5ffHFF6ZgjQZFa8VS93246fw+uzvuuCNoBVHrnEEgXPT8AQAAIEWunaxBKlglT/u2evXqhXUfOkfPStcItA8drVu3rtfi8fbbWIdz6gLvVlrQxVo9VHv6rG3UXkYd3modvhpXxdBs2bKF+OgAX/T8AQAAIMVp3769V0EXXXZh9uzZXsskLFmyxOs2Dz/8sE+RlKefftpz0p48K3dVTjcd4hkbG+sVvObNmxf0Nj/99JNpm532Ej7wwANy48YNz7aePXt6zfnTCqbNmjXzup21QqgWd9HHYFWnTh2f+wJCRc8fAAAAUhydv6fVMocOHerZpmFKi6DoMgwaBK2VL7t06eIz7FPDoRZqcStRooRX71zz5s299tcQV6ZMGbPcg/YAahEY6zp//m6jYW3KlClyyy23mPvXOX5a9EVva22fzgt89dVXfR7nkCFDzJxAd++fFnpZvXq1WUZC5xhu2bLFs68+bnuBGSAchD8AAACkSDr3TufOuStnak/cd99957NfjRo15IMPPgj7+DrMU5dlmDBhgmebLrWgYc4fnf9nr8BpHeJpX7fPrVixYvLjjz/6LfDSpEkTE3Bffvllz7Zff/3VnKx0eKgWl9FwCsQXwz4BAACQIrl7+DQEZs2a1ed6HYLZp08f08OnPW7xofMKtcJmwYIFA+6TO3duE8600qZ9CQqtDKpr+/mjQzx1uKkOUa1YsWLA42txGR3eqT2T/lSvXl0WLVok3bt3D/lxAf6kcvlbNAUJmpystLseAAAAkXH58mVZuXKl/Pvvv2YeXeHChaVhw4Z+Q6GbDqfcvHmz17p8GqT80X+JdV9daP3UqVPmsvbU6VBOXbIhbdrgA+b+/vtv2bFjhxw6dMiE0qJFi5riLvY5gsHoff7++++mDbrIvQZabW+4VUyBQAh/EUb4AwAAAJASMewTAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4QNrkbgAAAEBKV3jAkeRuAoBkdGB4zE3x/NPzBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA6/wBSeT8+fNy6NAhSZUqlRQsWFAyZ84c8fs4cuSInDp1Slwul+TJk0fy588f8m0vXrxo2nfhwgXJmjWrFC5cWNKlSxfy7a9duyaHDx+WM2fOSJYsWcK+PQAAABIXPX9AIps9e7Y0atRIsmfPLmXLlpUyZcqYn5s2bSrz5s1L8PF37Nghjz/+uOTNm1cKFCggFStWlEqVKklMTIzky5dPevToIbt27fK53aVLl2Tq1KnSs2dP0yYNbKVLl5aqVatKyZIlTRtbtWolixYtCnjfq1atkr59+0q5cuUkffr0UrRoUalSpYq5faZMmaROnToyatQouXLlSoIfJwAAABImlUu7CBAx9evXN+erV6/mWXU4/dN69tln5aOPPgq634ABA+S9996L1338+uuvcvfdd5texWCyZcsmCxculHr16nm2bdu2zQTFULz88ssydOhQn+0aaleuXBnn7XW/xYsXS4YMGUK6PwBIaQoPOJLcTQCQjA4Mj7kpnn96/oBEoqHPHvyKFClihkNaDR8+XD7//PN43Uf37t19gp/eh56szp07Z3r4gtFhqNqDp0HRbtiwYbJgwYKgt8+VK5dUqFBBcufO7XPdihUrZMyYMXE8GgAAACQmwh+QCDRsaW+Z1ejRo2Xfvn2yf/9+E/isXnzxRTPnLhx///23GfJpNWXKFHMfeho3bpzXdZs3bzbb/fVW69BUnSu4fft2OX36tIwYMcJnP/vxVLNmzeTLL780tzl58qRs3bpVTpw4YY5nn9P4888/h/X4AAAAEFmEPyARfPPNNyYAummP2NNPP+253L9/fylRooTnsgav77//Pqz7iI2N9bqsRWQ6d+7suaxz/ey9eNbbaDgbP368mbfXpk0bM2dPpU6dWvr162e2WWmws9OhoF26dJEcOXJ4bdfb3nfffV7bbty4EdbjAwAAQGQR/oBEsGzZMq/LTZo08dmncePGQW8Tl1KlSknatP9XsFfD5uXLl70uW3sTdb6dNXAWK1bMFIoJpHr16l6XtYBLOOy9jLVq1Qrr9gAAAIgslnoAEoEOsbQqXry4zz72bfbbxCVnzpwmvI0dO9bTq/fQQw+ZIjPay6ZFZHT5Bbc+ffqEFeB0WKlV3bp1gwY9He6pVT0PHDggX331lSlG46bVP59//vmwHh8AAAAii/AHJAKd/2alyybY2Ydk2m8TCi0oo71/Oh/v6tWrMmvWLHOy0h4/DX7vvvtuyMf9/fffZebMmZ7Lul6fddiq3euvv26GkNplzJhRevfuLS+99JLfQjAAAABIOgz7BBKBrqFnlSZNGp99rEM2/d0mFBrKHn74YVN4JZDbb7/d9Aj6a4M/ugRE69atTZh00wI1oS4LYX9MS5YskbVr14Z9WwAAAEQW4Q9IBPZePetcvEDb/PUOxkV73HQNPfdC7FqsRef16ZDSVKlSmW1z58416/vZK4z6o0GtQYMGcvDgQc+2wYMHm6Gkwej8wWrVqpnCNrrkg9XGjRulbdu28u2334b9+AAAABA5hD8gERQqVMjr8vHjx332sW/Tap3hWL16tQwZMsQsJq90/cAtW7bI7t27Zc+ePWboZp48ecx1Ogdw4MCBsmnTpoDH0yUbWrVqZSqPKg2POlT0jTfeiLMtr776qvzxxx+mIqgOX12/fr0Jg256/7qYPQAAAJIP4Q9IBPbKlv6WSbBvq1mzZlj3MW/ePK/LjzzyiJQvX95zWcNXhw4dvAJYoIXaNbx169bNM9RT5+pNnTpVXnjhBYnv47f3NP77779y7NixeB0PAAAACUfBFyAR6Dp377//vufy4sWL5cyZM5718LQypn1pBx0aaaW9d7p4unVopbVoiruHzu3w4cM+7Th06JDXZfttdOipVgzVoOeWL18+UzRGh38Go0tJ2Ie3Wu3atctn2/Xr14MeEwAAAImH8AckAi3Acsstt3iGWZ49e1buvfdeeeWVV0wPnM7Vu3Dhgmf/W2+91czLs9JhkjNmzPBcnjBhgumdc6tUqZLX/pMnTzZDP3Xopg4FnT17tjlZWW+jbbrnnntkxYoVXstHfPrpp2YBeB3GaaXDQK1DOT/88EP5/PPPpV27dqbXUu9b5y1q757OQRwzZozX7XUeYkxMTBjPIgAAACKJ8AckAg1KuvSBLu7uXmh96dKl5mSXJUsWs1RDuLSCp875O3r0qLmsofKtt94yJ3+KFi1qAqh1HT9r8FPa09i+fXu/t9dqodZ1A9XevXtl1KhRIbV32LBhniI0AAAASHrM+QMSSe3ateWnn37yKf5iD2QLFy40vYTh0qqaOu+vVKlSce6rcwHnz58vWbNmlUhxD2GNiw4j1V5JXZICAAAAyYeePyCRh39u375dpk2bZoZCHjhwwBP6dP29zp07S6ZMmfzetmTJkl7DLP0tkq7DLbVwzJw5c+SXX34xvXnae6c9bBrOdOmFli1bmuGd9nUFdWin9fhxsd/+mWeekfvvv98E3HXr1pn5hUeOHDE9hBpMy5UrZ5ahuPvuu819AQAAIHmlcrnrxCMi6tev7ynDDwAAbg6FBxxJ7iYASEYHht8cdQsY9gkAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAABwgbXI3AIlv3JfTeJoBB+vZtVNyNwEAAKQA9PwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcICoC3+7du2Sjh07Sq5cuSRdunRSqVIl+eijj8TlcoV1nEWLFkmLFi0kJiZGMmbMKBUqVJABAwbIiRMnEq3tAAAAAJBcoqra586dO6VevXpy/Phxz7atW7fKM888Y657//33QzrOBx98IH379vXatn37dnP67rvvZMOGDZInT56Itx8AAAAAkktU9fz169fPBL/WrVvL3r175cqVKyasZcmSRT788EP57bff4jzG5cuXZfDgweZnPT969KjZtnr1aqlcubI57scff5wEjwYAAAAAkk7UhL9jx47J3LlzJW/evDJ9+nQpVqyYGfbZvn17ef31182wz/Hjx8d5nH379klsbKzUqFFD3njjDcmXL5+kT5/e9CiOGDHC7LNly5YkeEQAAAAAkHSiJvytXLlSbty4IW3btjU9fVYPP/ywOV++fHmcx9E5fqlTpzbHsnNvK1SoUMTaDQAAAAApQdSEvx07dpjzKlWq+FxXoEAB0yPo3ieYbNmySbdu3WTjxo0ycOBA0xN47tw5Wbp0qTz//POm+EvPnj0T5TEAAAAAQHKJmoIvZ8+eNeda5dOf3Llzm/mAV69eNcNBg/n000+lYMGCMnr0aHn33Xc923Xo58SJE6VixYpxtqd+/fp+t2/evNlvQAUAAACA5BQ1PX9xCWeph3/++UeWLVvmCZRu27Ztk/nz5/sdEgoAAAAA0Sxqev5y5MhhzgOtw3fy5EnJlClTnL1+ul/Tpk3l4sWLppevTZs2Zijo33//bap/ahGY69evy9ChQ4MeR6uDhtMjCAAAAADJKWp6/sqWLesZVml38OBBEwrd+wQza9YsMzxUl43o2rWrGS6qgVGXefj6669NMZnPP/88UR4DAAAAACSXqAl/DRo0kDRp0sjs2bNNgRaryZMnm3Pt0YvLqVOnzLm/oZ3a46fDR7V3EAAAAABuJlET/rSaZ7t27Uww69Chgxmmqev1TZ061azzlypVKnn88cd9wty1a9e8ttWsWdOcjxw5Uj777DM5dOiQOc6GDRvk/vvvlwsXLkitWrWS9LEBAAAAQGKLmvDnDmz58+eXhQsXSvny5c1cPV3jT+fvvfDCC1K9enWv/Vu0aGGGdK5fv96zrXnz5maen96md+/eZk0/PU7t2rVl3rx5kiFDBnnnnXeS4dEBAAAAQOKJqvBXvHhxWbdunTz66KNmsXadn6eBb9y4cX4Dmw4T1ZP2ClrNnDlTPvzwQ6lbt64pJKNr+xUtWlQ6d+4sa9eulSZNmiThowIAAACAxBc11T7dihUrJpMmTQpp359//tnv9rRp08qzzz5rTgAAAADgBFHV8wcAAAAAiB/CHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcADCHwAAAAA4AOEPAAAAAByA8AcAAAAADkD4AwAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcICoDn/Xrl2L2LGuX78esWMBAAAAQEoTdeFv27Zt0q5dO8mcObOkT59eSpUqJcOHD5cbN26EfawffvhBmjZtKhkzZjSnatWqybhx4yIaKgEAAAAgJUgrUeTvv/+WBg0ayKlTp8zltGnTyu7du+WFF14w5x9//HHIxxo4cKC8++67nsvp0qWTTZs2yRNPPCE1atSQ2rVrJ8pjAAAAAIDkEFU9f3379jXB7/7775fDhw/LlStXZO7cuZI9e3b55JNPZO3atSEdZ8aMGSb4aW/fRx99JKdPnzbH0nDZu3dv06MIAAAAADeTqAl/R44ckfnz50v+/Pnlq6++kpiYGEmVKpXcc8898sYbb5h9JkyYENKxXnvtNXM+duxY6dOnj+TIkcNcLlu2rIwZM0ZuueWWRHwkAAAAAJD0oib8rVy50szra9OmjWTKlMnruk6dOpnz5cuXx3mc7du3y19//SXFixeXRx55xGxjjh8AAACAm13UhL+dO3ea8ypVqvhcp72B+fLl8+wTzG+//WbOW7ZsKcuWLZPq1aubYZ5ZsmSRtm3byp9//pkIrQcAAACAm6zgy7lz5+Tzzz+XX375RY4ePSpVq1aVESNGyLRp0yRXrlzSsWPHeB9X6TH8yZ07txw7dszM3Qs2Z+/EiRPmXOf5tWrVyuyfJk0auXDhgsyZM8e0e8WKFSYUBlO/fn2/2zdv3uw3oAIAAADATdPzt2/fPlMps1+/fmZ+nhZg0aUZtCCLBkAdnrl3794E3YfL5fK73b3Ug84DDOX2WvRF5wtqldCrV6+aXsPWrVvL+fPnZcCAAQlqIwAAAADc1OGvR48eJkT1799fFi1a5Nmugezhhx82wUuLtcRHzpw5zfnx48cD9ujp2n+6ZEMox9GewilTpkiJEiVM+3S9wOnTp5ugqsNBL1++HPQ4q1ev9nui1w8AAADATR3+dIinBj6tnDls2DCfEFahQgVzvm7dungdXytxKl2Lz27//v1y8uRJzz7BlCtXzpxr2LMXjtF5fyVLljQFYM6cOROvdgIAAADATR3+9uzZY3r2ypQpIxkyZPAZfqkFWVR8Q1XDhg3Nou46L89+jC+//NKcN2vWLM7j6LBU7f3TNf3sx9EAuWPHDrP+X6C5hQAAAADg6PCnQy5VbGxs0EIrefLkidfxdZimLu6uhVruvfde0wOoQ0C/+OILefPNNyV16tRm2KnVxYsXTXvc8wGVFoN5/PHH5ezZs+Y4a9asMYViVq1aZap96pw/nfsX1/BRAAAAAHBktc/y5ctL1qxZTc/ZwYMHfXr+li5das7r1q0b7/vQojG6lp8eq1q1al7XDR482Ge+3V133WXm7+lQ09q1a3st8r548WJzHHvVzsKFC8vIkSPj3UYAAAAAuKl7/rSn7IknnpDr169Lt27d5N9///Vcp5U/J0yYYIqpdO3aNd73UaRIEVm/fr307NnTLNKuvYj16tWTyZMnyxtvvOGzv87p03l8upSDlbZDl3MYNGiQmYuoQzx1vuAzzzwjGzZskKJFi8a7jQAAAACQEqVyBVo7IR4uXbok7dq1k4ULF/pcp/Povv32WzOk8mbm7knUyp8pxbgvpyV3EwAko55dO/H8AwlUeMARnkPAwQ4Mj5GbQUQXedeAN2/ePLNkwsyZM80aerqtVq1aplctlGqcAAAAAIAUHP60+Iou9ZA3b17p3LmzOQEAAAAAbrI5f1roRQPf66+/HqlDAgAAAABSWvjTKpnutfIAAAAAAClLxMKfVsjUhdi1B3Dv3r2ROiwAAAAAICWFPzV16lSpXLmyWYxdF03X6p8AAAAAgJuo4MvKlSuladOm5mdd6097AXWh99SpvfNlo0aNPAu+AwAAAACiLPzlyJFDmjVrFud+VatWjdRdAgAAAACSOvxVqVJFFi9eHKnDAQAAAABS6pw/AAAAAIDDwt/Vq1fl0KFDcurUqcS6CwAAAABAcoW/jRs3yj333CPZsmWTQoUKSe7cuc0yEEOHDjWBEAAAAAAQxXP+1Jo1a+S2226TixcvSkxMjFSrVs30/P32228yePBgc/3s2bN9KoACAAAAABJXRFPYU089ZYJf165dZc+ePbJgwQJZu3atOeXKlUt+/PFHmTFjRiTvEgAAAACQlOFv//798vvvv0vWrFnlk08+kYwZM3quq1mzpgwaNMj8rD1/AAAAAIAoDX8HDx4052XKlJHMmTP7XK9DQNWBAwcidZcAAAAAgKQOf9mzZzfnhw8fDhoOdTF4AAAAAECUhr9y5cpJ/vz5TfibMGGC13U6D/D99983Pzdq1ChSdwkAAAAASOpqn1rBc8iQIaboS/fu3WXhwoVSt25dU+3zq6++kl27dkmxYsWkZ8+ekbpLAAAAAEByLPXw5JNPyqVLl+S1116T6dOnm5NbvXr1ZPLkyZ7hoQAAAACAKA1/qm/fvtKjRw9ZuXKlmeenVT9vueUWqVKlSqTvCgAAAACQXOFPZcuWTe68887EODQAAAAAILkXede5fdWrV5cRI0Z4bT937pzUr19fmjRpIpcvX47kXQIAAAAAkjL8uVwueemll2TTpk3y4IMP+vQE1qpVS5YvX+5TCRQAAAAAEEXhb+/evbJv3z4pUaKEFC1a1Of6xo0bm/MVK1ZE6i4BAAAAAEkd/o4dO+bp5fPHvf3QoUORuksAAAAAQFKHv8KFC5tzXc/P37y+v/76y5wXLFgwUncJAAAAAEjq8FeoUCGpWrWqxMbGynvvved13YkTJ+TDDz80P991112RuksAAAAAQHIs9fD2229LmzZtZPDgwbJ+/XpT3VOD35dffikHDhyQ2rVrS8eOHSN5lwAAAACApA5/99xzj1nu4bnnnpMffvjBnKzXTZw4UdKmTZSlBQEAAAAAQUQ8iXXu3Fnuu+8+Wblypan+mSlTJrPMQ9myZSN9VwAAAACAECVKN5wGvpYtWybGoQEAAAAA8ZCoYzCXLFkiq1evNgvA33nnnaYHEAAAAAAQZdU+f/75ZxPqhg8f7rVdw163bt3ktttuk5dfflleeeUVqVOnjikIAwAAAACIsvD3ySefyIIFC3zm802dOtVU+KxYsaL897//lfvvv98EQg2BW7duTWibAQAAAABJFf6uXbsmP/74o6RPn17uvvtur+vGjx8vGTJkkEWLFsnAgQNlxowZ0qpVK7lx44ZMnz49vncJAAAAAEjq8KeVPC9fviwlSpSQdOnSeYVCnefXtGlTKVy4sGd7p06dzPmmTZvie5cAAAAAgKQOf0ePHjXnWbNm9dq+bds2uXTpktSrV89ruzsInj59Or53CQAAAABI6vCXJ08ec75nzx65fv26Z/uqVavMub2y55kzZ7xuBwAAAACIgvBXunRpiYmJkZMnT8rEiRPNNg2BWuhF5wE2adLEZ5ioKl68eELbDAAAAABIqvCXKlUq6d+/v/n5iSeekIYNG0rlypVNz1+XLl0kZ86cXvsvW7bMnDdv3jy+dwkAAAAASI5F3gcMGGB6/kaOHOkZ7vnAAw+Yy1bnzp2T+fPnS/bs2eX2229PyF0CAAAAAJI6/Gnvny7c/tJLL8nevXulSJEiPj1+SoeB/vnnn5IxY0azBAQAAAAAIErCX2xsrKnsmS1bNilfvrxUqVIl4L4a+MqUKRPfuwIAAAAAJNecvz/++EPq1Kkj3bt3T2gbAAAAAAApNfwBAAAAAKIH4Q8AAAAAHIDwBwAAAAAOkKBqn+rUqVMyd+7ckPfPnTu3NGjQIKF3CwAAAABIyvC3ZcsWadOmTcj762LwK1asSOjdAgAAAACSMvwVKFBA2rVrF/L+LPkAAAAAAFEY/kqXLi2ffvppZFoDAAAAAEgUFHwBAAAAAAcg/AEAAACAAyRp+Dtz5ox8++23SXmXAAAAAICEhL9ixYrJyy+/LF27do1z35MnT8qrr74qxYsXlw8//JAnHgAAAACipeCLhr+hQ4eaZRu6dOkie/fulZiYGHnsscfkrrvu8qwBOHz4cBk9erScO3dOsmfPLu3bt49k+wEAAAAAiV3tc9asWSbM3bhxw7NNh3VOmjRJSpQoIQ8++KAcPnzYLOz++uuvy7PPPis5c+ZMyF0CAAAAAJI6/L344osm+Oki7xrs9uzZI/379zfDQWNjY+X8+fPm50GDBknWrFkTclcAAAAAgOQIf/v375d//vlHsmTJIlOmTJFs2bKZ7drTN3jwYE/PYDgLwAMAAAAAUljBl4MHD5rzsmXLeoKfql27tjkvVaoUwQ8AAAAAoj38XblyxZxrz5+Ve3hnwYIFE9o2AAAAAECEsMg7AAAAADhAggq+qC1btkjr1q09l3V5B3/b3SpXrizvvPNOQu8WAAAAAJCU4U/D3o8//hjy9tOnTyf0LgEAAAAASRX+atasKX/++WfYt7PPEQQAAAAApODwlzlzZqlSpUpkWwMAAAAASBQUfAEAAAAAB4h3+Nu5c6f07t1bhg8f7jPXb+7cubJq1Sqv7TpEtFGjRvLUU0/Fv7UAAAAAgKQNf4cOHZLPPvtMZs2a5bX9r7/+kjZt2siLL77otf3MmTOycuVK2bRpU3zvEgAAAAAQTwz7BAAAAAAHIPwBAAAAgAMQ/gAAAADAAQh/AAAAAOAAhD8AAAAAcIB4L/LutmXLFmndurXXUg/BtgMAAAAAojD8aaj78ccfQ94OAAAAAIii8FezZk2zcHu4smTJEt+7BAAAAAAkdfjLnDmzVKlSJb43BwAAAAAkIQq+AAAAAIADJHjOn3K5XHL+/HnJmjWr1/aFCxeaeX96XbNmzaRz586SOjV5EwAAAACSWoKS2LVr16R///6SI0cOyZYtm+TOnVtef/11c52et2rVSkaNGiXjx4+XRx99VDp16hSpdsv169dNqIyEGzduyPHjx83p6tWrETkmAAAAANw04e+tt96SkSNHSmxsrJQpU8b0AA4ZMkTeffddeeONN6ROnTryxRdfyODBgyVdunTyzTffyIIFCxLUYC0yo6EyY8aMpqexUKFCMnToUBNE42v48OGSL18+c1q2bFmC2gcAAAAAN9WwT+150+Cnpk+fLg8++KBcuHBB7rjjDnnllVckTZo0MmfOHImJiTH76HUjRoyQWbNmmfAWH7p2YKNGjeTs2bNm+KhWDj106JAJl3v37pVx48aFfczdu3ebXsqcOXPK6dOn49UuAAAAALhpe/727dsnZ86ckaJFi5rg564A+uSTT5qhk9oT6A5+qnHjxuZ8165d8W5s3759TfB76KGHzBBN7XFcvHixCW6ff/65rF69OuxjanurVq0qDzzwQLzbBQAAAAA3bfjTHjdVuHBhr+1FihQx5zr/zypPnjyeHsD43t+iRYukQIECMnHiRMmVK5fZ3qJFCzPsU02YMCGsY06ZMkV++eUXExwpRAMAAADgZpY6IcM+lQ7vtLJfttN5gfGxatUqc9s2bdpIhgwZvK5z9zyuXLky5OOdPHnS9CQOGjSI9QoBAAAA3PSiZt0F93DRSpUq+VynhVry588vO3fuDPl4WqVUeyNffvnliLYTAAAAAG7Kdf60CEvr1q09l0+dOhV0e3zpXD/lHu5pp9uPHj0qV65ckfTp0wc91tKlS2XSpEny66+/+vQihqp+/fp+t2/evJmeRAAAAAA3X/jTUKcLuYe6Pb5SpUrlWZPPH/f2uObuXb58WXr16iW9e/eWhg0bRqx9AAAAAHBThr+aNWuaNffCpcszxIdW9FRa5dMf3a7HTps2+EN68803TTAdMGCA17EuXbrk6WHU7Xp/wY4VqLJooB5BAAAAAIjK8KfLOlSpUkWSSrly5cz5xo0bfa77999/TaCrUaNGnMf5+uuv5dixY1KqVCm/17dv396cL1++3KwpCAAAAAA3gwQP+0wqOkRTe+J04Xit1GldSuKLL74w582bN4/zODo30L3shNX58+dN71/27NklXbp05gQAAAAAN4uoqfapoU2XdNBhmW3btpW1a9fK/v375ZNPPpG3337bLDHRo0cPr9voIvQ6hPPatWuebXo73WY/de3a1Vw/Y8YMc7lu3bpJ/hgBAAAAQJze86eGDx9uhmPqen72cKZz+SpWrOi1rV27drJs2TJZt26d1K5dO4lbCwAAAAApR1SFv4IFC8qGDRtM0Fu4cKHExsZK2bJlpU+fPtKhQwef/XPkyGGGeIYyhDNr1qxm37iWiQAAAACAaJTK5XK5krsRNxN3tc9A1UCTw7gvpyV3EwAko55dO/H8AwlUeMARnkPAwQ4Mj5GbQdTM+QMAAAAAxB/hDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOADhDwAAAAAcgPAHAAAAAA5A+AMAAAAAByD8AQAAAIADEP4AAAAAwAEIfwAAAADgAIQ/AAAAAHAAwh8AAAAAOEDUhr+rV6/KmTNnEnwcPcalS5ci0iYAAAAASKmiLvz98ccf0qJFC8mYMaPkzJlTYmJiZMiQISYMhuL8+fMyceJEueOOO8zt9ZQ5c2apWrWqjB07VlwuV6I/BgAAAABIamklimzevFmaNGki586dk7Rp00rWrFnl6NGj8vrrr8vevXtlwoQJcR5jyZIl8thjj3kuZ8+e3RxPj92rVy/ZunWrvP/++4n8SAAAAAAgaUVVz1+/fv1MUHv44Yfl+PHjZsjm0qVLJVeuXKY3b+XKlXEeQwNjt27dZNGiRXLq1ClzjNjYWHnjjTfM9R9++KHs2bMnCR4NAAAAACSdqAl/Bw8elMWLF0vBggVl/PjxkiNHDrO9adOmMmzYMPOzBsC4NGvWzPQQtmzZ0gz5VDrsc/DgwXLnnXeaYZ/aCwgAAAAAN5OoCX+rVq0ywaxNmzaSIUMGr+seeOABcx5Kz18wRYsWNef58uVL0HEAAAAAIKWJmvC3a9cuc16pUiWf6/LmzSv58+eXnTt3xvv4p0+flh9++EEqVKggderUSVBbAQAAACCliZqCLzrXT7mHatrpvD8t/nLlyhVJnz59WMe+ceOGdO3a1cwB/P777yV16rgzcf369f1u1yGjVapUCev+AQAAACCxRU3PnzuQaVDzx709lOBmde3aNenSpYv8+OOPMnnyZGnQoEEEWgsAAAAAKUvU9Py5e/y0yqc/uj1LlixmCYhQ6eLuHTt2lJ9++kmmTJlifg7V6tWrw+oRBAAAAIDkFDXhr1y5cp5F3u10jT8dslmzZs2Qj3f27Flp27atKRIzbdo06dChQ0TbCwAAAAApSdQM+2zYsKGkS5dO5syZIydOnPC6Tpd+UM2bNw/pWDo3UJd80Aqi3377LcEPAAAAwE0vdTQN+9RhmVr4pXXr1ia46WLso0aNkv/+97+SJk0a6dGjh9dtTp48KYcPHzbz+twOHDggTZo0kY0bN8onn3wi9erVM/tYTxcvXkyGRwgAAAAAiSdqhn2q9957T5YvXy5r1qwxPYFWb7/9tlmmwer++++XZcuWybp166R27dpm27x582T79u3m5549e/q9nzFjxkjv3r0T7XEAAAAAQFKLqvBXoEAB2bBhgwwbNkwWLlwosbGxUrZsWenTp4/ce++9Pvvnzp1bYmJizHBRt8yZM5ttweg+AAAAAHAziarwp/LkySMjR44Mad+ZM2f6bOvcubM5AQAAAICTRM2cPwAAAABA/BH+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAA0Rt+Dt//rwcOXJEbty4kSKOAwAAAAApWdSFv3Xr1knDhg0la9asUqBAAcmTJ48MGjRIrly5kizHAQAAAIBokFaiyMaNG6VZs2Zy4cIFyZgxo2TPnl2OHj0q77zzjuzbt0+mTJmSpMcBAAAAgGgRVT1/ffv2NYHt8ccfl+PHj5vhmmvWrJG8efPK1KlTZdmyZUl6HAAAAACIFlET/vbv3y9LliyRwoULy6effipZsmQx2+vWrStvvfWW+XnSpElJdhwAAAAAiCZRE/5Wr15tztu0aSPp0qXzuq59+/bmfNWqVUl2HAAAAACIJlEz52/37t3mvGLFij7X5c6dW2JiYmTXrl1Jdpz69ev73b5+/XozjzDQ9cnh6LHjyd0EAMnoi09H8fwDCXRs71WeQ8DB6q/07jRKblWqVJFx48bdvOHv3Llz5jxHjhx+r8+ZM6eZu3f58mXJkCFDoh8nkLRp00rmzJklJcmfL29yNwHJZPPmzZ43CABA/NUqnrL+8UPS4vMUN4uoCX+pU///EaqB1uO7fv26OU+TJk2SHMc9fBRIydw90Py+AgDA5ykQNXP+cuXKZc6PHTvm93qt2qlr9mnPW1IcBwAAAACiSdSEv3Llypnz33//3e88vtOnT0v58uWT7DgAAAAAEE2iJvw1atRI0qdPL3PnzjULsluNHTvWnLdo0SLJjgMAAAAA0SRqwl/27Nmlc+fOEhsbK3fffbf88ssvsm3bNnn33Xdl+PDhZtmGnj17et3m8OHDsmfPHrly5UqCjgMAAAAA0S6Vy+VySZTQeXoNGjSQHTt2+Fw3cuRI6du3r9e2Zs2aybJly2TdunVSu3bteB8HiFYUfAEAgM9TICrDn9I5edpLt3DhQtN7V7ZsWenTp4/ceeedPvs+9NBDsmbNGpkzZ45UrVo13scBAAAAgGgXdeEPAAAAAHATz/kDAAAAAMQf4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwApyoEDByRv3ryybt06c/no0aPSuHFjGTt2bHI3DQAAIKoR/gCkKIULF5ahQ4fKfffdJxkyZJDSpUtLmTJlpGvXrsndNAAAgKiWyuVyuZK7EQDgz9WrVyVdunQ8OQAAABFA+AMAAAAAB0ib3A0AAAAAwrVr1y45e/Zs0H102kDWrFn9zi8/c+aMFCxYUHLlyhXnfen9HDx4UPLlyyd58uTxu8+WLVvkypUrfq+rVKmSpE+f3mf7qVOn5NChQ5I9e3YpUqRInO0AEoo5f0AA+sHQr18/qVq1qnmjr1KlivTp00f2799vri9QoID5QInr5Hbu3DkZM2aMtGjRQooWLSr58+eXevXqyciRI83wRrsmTZpIiRIl5PLly/Kf//xHypYtaz50mjVrJgsXLvTaN9y2qN9++00eeughKV68uHl81atXlzfeeEPOnz/vty3W4+j++ry88MILcvLkSb/tjovO5dN97bfV7XZffvml5771wxcAgKeeekpq1KgR9LRmzRqvJ2rq1Knmc0aDVuXKlc3nmRYV+/333/0+oStWrDDXa0CsWLGiKUhWoUIF+e6773z2veOOOwK2w/7Zpcdt0KCBuX9th/5foO2aOXMmLywSl875A+Bt3bp1rty5c+t8WJ9Ts2bNzD5ZsmTxe7395NajR4+A+9x///0+L0GtWrVcefLkcbVt29Zn/9SpU7umT5/u2Tfctvz444+u9OnT+92nZs2artjYWJ+2BDqm+/mwtzsuMTExZl/7bXW71bFjx8zx3Pe3b98+fl0BAK5WrVqZz4Vq1ar5nAoVKmSuW7RokeeZGjdunOezJGvWrK4yZcq40qVLZy7r5+jmzZu9ntW5c+e60qZNa65PkyaNq2TJkq58+fKZy5UrV/Z5BfQ+M2bM6NWOvHnzmv13797t2W/BggWe42o7ypcvbz779HKqVKlcM2fO5NVFoqHnD7C5du2a6RHTHi39Fm/JkiVmSMZff/0lH3/8sWdYxpEjR0xvnvuk3wiWKlXKa5ue3HLkyCEvvviirFy5Ug4fPmxuv2rVKrnzzjvNN32bNm3yeS1OnDhhvrX86quvTI+jDinp37+/3LhxQ5588klPL104bbl06ZI8/vjjZmhKt27dZOPGjebxzZkzxwyP0R7BYcOG+bRFv510H+f48ePmeSlWrJgsW7bMb89lpAwYMMA8zpYtWybafQAAotcff/zhc+rbt6/XPhcvXpSBAwean4cPH26GfP7zzz/m80w/8/Vzxn290s+1Xr16mf8Jnn/+efM/gQ4z1eWHtm/fLg888IBPO/RzVXsFre3o2LGj1z56vCeeeELSpk1rRrXosM9t27aZ/wsWL15shn/qqBogsTDnD7DRMLNz505p2LCh/PTTT5ImTRrP0Eods6/DTFSWLFm8bpc6dWpJlSqV37kF6u2335bPPvtMBg0aZD5wYmNjtStOrl+/bq7XD4lbbrnF53bjx4+X1q1be5ZB0A8tHZI6ffp0WbRokdx7771htWXp0qUmLN5+++0yYcIEz3a9Dw1/Opxz2rRp8tZbb/nc1n08PY+JiTFLMej+iVWRU9uqH4669MO+ffvMByMAAOHSL1I1wLVt29Z8ieqmYeuLL74wX2jqZ6oGOJ2bp/vrZ61+Vr7//vtexypXrpy89tprPvehn+v+5vVZrV27Vvbu3SsdOnSQmjVrytatW812/X9Ap3bcfffd5jN49+7dUrJkSV5oRBw9f4CN9vAp/bbOHfwiQY/3zDPPyPLly803fPohod80ak+cunDhgs9t9JvBVq1a+Wxv06aNOf/777/Dbod+Y6l0HT07/cZST3v27PHpzdNeSPe8O/1w0yCsYVQDsp11X50nofvqOn3+ejcD0Q/g3r17m15M7TEFACC+9AtEpV/s2mXKlMkEMf3c0S9HlX4OKp2nHwotCKOf54G+AHbT3kOlcwb1y1P90ldP1apVMycNfsrdDiDS6PkDAv1xpI3cn8eGDRvk+++/N0VbtMCLvtHrMFANl9988410797d7+20B09PgdqWkGU6AwVb97F1aKmdvRiM9szp0NG5c+eaXsBA+54+fdp8u6mPVYeVaqCLi/aUarjVnljW+gMAJIT7M8o6HcPKXTXUvZ/7XD+/wvniuFChQkH3c3+e6ZenWjwmkIwZM4Z0v0C46PkDbDSgqdmzZ0fsuXH30Ok4fh1eqXPlNPzpN4Ra8SsQ/RZSewrtdGiK0nl94XIPI1mwYIHPdTqvUEOafijZw5x1zp9+GOr8w4cfftgMxbQOHw22r34r+u2334b0fGn407mJWmUNAICE0Iqa6uuvvzZVtK10zt26detMFW49Ka3Q6a4O6i8AukftuLmrf2rvXTDaw6huvfVWv3MV9aS1AbQCN5AYCH+AzW233Wa+uZs/f77p1dJhkjovTydla2GW5557LuznTEs4qylTppix/kqHfr7yyis+wcnusccek19++cUMw9QPIJ3zp/MTdJ6fzkWIz+PLmTOneSx6/8eOHTOPT3sndSiofii2b9/e723dQzk1uOpz5F7ryD2cJpR9dbJ7XLSYTbZs2eTdd98N+/EBAGCnyzXVqVPHzLlv2rSp+SJSi67pXHwd2qlftuoXjtYvgnXaxb///mtu9+mnn5ovY+fNmycvvfSSZziojmZ59tln5YMPPjBz7du1axf0ydfj6rw+HQ2kBd+0oJt+CazHnjRpknTp0sXv0FQgYhKvkCgQvRYuXGjKNbtLQmuJZ/fPTZs29XsbLftcunRpv9ddvnzZXO8+hrvEs546dOhgzseMGeOz7EGuXLlcjRs39mmDnj7++OOA7Q/WFjV58mRTTtrf49PbnThxwqct7lLYerI+N7rsxOrVq0PaV+9nw4YNQZd6cO/75Zdfel3Xq1cvlnoAAPgs9eDPe++957PUw9atW10FCxb0u2xRkyZNXBcvXvQ6xqFDh7w+u62n2rVrm31atGjh2TZkyBCfdvTp08dnqYfjx4+7br311oBLKNk/G4FIoucP8EN71P73v/+Zalzac6U9Y1qFS0s7jx49OuznTAuk6Lw4rTKmvWE6n06HoOjEbh0OGYjO99Php48++qhn/H/58uXNt4PuqqPx8cgjj5ieTV1UXdumj0+HumiBFa1wljt3br+303l87iI12nuoS2HoEFRdrD7Ufd1DXoJp3ry5+fYTAIBAdFH0QMMs9TNNr9NRJG5a0EwLj73++uum50579PRzedy4cWYKg32enVb5Xr9+vYwZM8b06OlQTd3/nXfeMSNylM7h16Ubfv31V78VQHV5KG2HtQqojoTRoZ06GqhTp05St25d0xupo410mKn2SAKJJZUmwEQ7OnCT0CGXcRUd0TWE9M8pc+bMcR5Pw5a74Ir+rLfVDx1rkZnatWubamO6BlE47Qi3LbqfHjdYeWo9nntJCqVtsM8JDHdfrW6qQ2S0ypr9trq//XHqcFRtpw531dsBAAAgPFT7BEIQSuCyhphwKm3qz3GVhg6nHeG2RYNUXOsShXO8UPf1F0yD3VYDYaDACQAAgLgx7BMAAAAAHIDwBwAAAAAOwJw/IIXS+W9aGEbnuAEAAAAJRfgDAAAAAAdg2CcAAAAAOADhDwAAAAAcgPAHAAAAAA7AOn8AAACISi6XS2bPni0//PCDbN++XWJjY6VAgQJSrFgxad26tdx9990hr5FrN3ToUPnqq6+C7jNr1iypUKFCPFsPJD3CHwAAAKLOvn37pH379rJu3Tqv7Zs2bTLnn3/+uTRq1EiWL18er+MfPnzYBMpgLl26FK9jA8mF8AcAAICocvr0aWnevLns3LlTChUqJH369DFBL2fOnHLkyBETDOfOnWt+TqgxY8ZIs2bN/F5XsmTJBB8fSEqEPwAAAESV1157zQS/WrVqycKFCyV37tw++zz++OMR6ZkrUqQIQztx0yD8AYlg48aN8v777wfdp2bNmvLss8967f/oo49K3bp1zRwDHbaSLVs2adeunTRo0MDvMf755x+ZMWOG7NmzRzJmzCh16tSRDh06SIYMGQK256mnnpJbb73V63r9cHz66afl2rVr5pvUrl27hv045s+fL9OnT4/zubn33nvNCQCA+Lhy5YqMHz9eUqVKJV9++aXf4Oemn41W2hM4atQo+fXXX+XMmTOm17BNmzbSs2dPSZ8+fbxfkF69epnhpUuXLpX8+fP7XN+jRw9ZvXq12cfd3sRqCxCUC0DEzZkzx6V/XsFO7dq189l/yJAhrooVK/rsO3DgQJ/7GDFihCtNmjQ++5YuXdq1ffv2gO255557fI41duxYz/W9evWK1+N477334txXT6+99lqEn20AgJOsXr3afJ5Ur149rNv9+eefrvz58/v9bKpfv77r/PnzXvv36dPHXKefhXGZMmWK2Vc/C+0OHjzoSps2revuu++Od1uASKHnD0hEjzzyiLRo0cJr29WrV+WJJ57wu//w4cMla9as8sorr0hMTIxs2LBBJk2aJO+88440bNjQfCOofv75Z+nfv7/5+YEHHjDzHHT+g/YYam/g/fffb3rt0qRJ43X8pk2byk8//WQmsJcvX95TKe2DDz4w8xn0G8v4Po677rpL8ubN67m8fv16+fjjj80+9evX92yvXr16yM8fAAB2+/fvN+fhVtnUUS1Hjx41n6cvvviiFC5c2Iyy0SGk2iun5++9957P7Z588kkZMGCAz3atKur+3NTCM88884zpkbTvq72TOrKme/fuCW4LkGARi5EAfHrMRo8e7fOsXLx4MWDPX968eV1Hjx712n/y5MnmujvuuMOzrU2bNmbbyJEjfY59yy23mOvmzZvnc/wPPvjAVbJkSa/evZ9++slcN3v27IA9f6E+Dqtvv/3WXK/tBwAgUqZOnWo+Xx599NGQb7Np0yZzmwoVKrguX77sdZ2OlsmQIYP5DL5x44ZPz1+gU+HChb2O88wzz5jtK1as8NpetmxZ08t35cqVeLcFiBQWeQdSEJ3zly9fPp9eN/120VrKWnvVMmfOLM8995zP3Ibnn3/e/Gwvfa20J1Bvo72JJ06cMNtGjhwpt99+u1StWjWRHhUAAJHj/pw8cOBAyLf566+/PKNl7PPpypUrZ+bMHz9+3PTG+av2uXXrVp/TsmXLvPbTuXpKe//cdB8dkdOlSxfPeoMJaQuQUIQ/IAXRkOdPwYIFzcK1bufOnTPDQlOn9v0T1gnj7n380WEn+mGjH2Z//vmnLF68WPr16xexxwAAQGLSQmNa7GXNmjVmykModKqC0qkV/ri3azGZQNU+7afSpUt77adfompw++abbzyfwe4gaB3ymZC2AAlF+ANSEP0m0e7y5cumnLW1R1AriekaRmfPnvXZ3/2Nor9qY+4PFf12Uufj/fe//5VKlSpJq1atIvo4AABILFotUz+3Lly4YObIh6Jo0aLm3N+C73qc3377zVTK1i9WE0Krep4/f16+/vprU8Hzu+++M/PyrfMTk6otgD+EPyAFmTJliin57Hb9+nUzEVxDni7B4KY/6+RxXZ7B+s2gFnLR4jDufQLRpRl0SMnUqVNNr59+gwoAQLR4++23TUDSLzJ1esS2bdu8rj958qRMnjzZUxytXr16piiZLvyuX3zqZ6h7P3fxFQ2UCV1ioVOnTpIlSxbT4zdt2jS5ePGiV69fUrYF8IfwB6Qg+m2ghrbGjRubuQD6TaGuAaQfAAMHDvTsp4FQP1z0g61MmTKmyljLli2lWrVq5kND19HThW+D3c+sWbNkwoQJ5kMTAIBoopWj9QvMTJkymS9OK1asaAKVDsXUES558uQx8+zc8/J0TvyIESPMz//5z3/MPsWKFTOjarR3TtfVdX956q/ap79hn3qaM2eO1756nAcffNAMSX3rrbc8l60S0hYgoVjqAUhBtEy0FmrRDzTr8BYNadaCLDohfN68eeYbwt27d5shoEp78Dp37iyfffZZnPd1zz33JNKjAAAg8emyRn/88YcJWdqLpoXM3MXMNAi2bt1aHn/8cc/+Ggb1i9PBgwebaRb62amF0HQpI13yKNDSEe6lJfzRoZ3+hn7q57YeX5c70gJtdvFtC5BQqbTkZ4KPAsDng0ILqejQDvsbuA7l1B47/ZbvtttuM9v0Q0vX8Bs9erQZyrllyxZTjEW//WvSpEnASeF6rLVr18qePXvMN4m1a9f2zCUItT1uWlBGv3HU9f/c6/KF+zis9u7dK0uWLDHtL1WqFL8hAIBEo//OHjt2zMy30+Jp2iMYjE590CkVOj8+0GfskSNH5NSpU0GPo0XWsmfP7rNdp2FomwJdH25bgEgh/AEpgD38AQAAAJHGnD8AAAAAcADCHwAAAAA4AMM+gRQglDl5AAAAQEIQ/gAAAADAARj2CQAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAADgA4Q8AAAAAHIDwBwAAAAAOQPgDAAAAAAcg/AEAAACAAxD+AAAAAMABCH8AAAAA4ACEPwAAAABwAMIfAAAAAMjN7/8BjZ+LHT98Cu8AAAAASUVORK5CYII=", + "text/plain": [ + "
" + ] + }, + "metadata": {} + } + ], + "source": [ + "GREY, BLUE = \"#9aa0a6\", \"#1a73e8\"\n", + "fig, ax = plt.subplots(figsize=(6.5, 4.4))\n", + "bars = ax.bar([\"\u0441\u0442\u0430\u0440\u0442\u043e\u0432\u044b\u0439\\n\u043f\u0440\u043e\u043c\u043f\u0442\", \"\u043f\u043e\u0441\u043b\u0435\\nCoEvo\"], [init_score, final_score],\n", + " color=[GREY, BLUE], width=0.55)\n", + "for b, v in zip(bars, [init_score, final_score]):\n", + " ax.text(b.get_x()+b.get_width()/2, v+0.008, f\"{v:.3f}\", ha=\"center\", va=\"bottom\",\n", + " fontsize=12, fontweight=\"bold\")\n", + "ax.set_ylim(0, max(init_score, final_score)+0.1)\n", + "ax.set_ylabel(\"BERTScore\")\n", + "ax.set_title(\"CoEvo: \u0431\u044b\u043b\u043e / \u0441\u0442\u0430\u043b\u043e (SQuAD v2)\", fontsize=12, fontweight=\"bold\")\n", + "ax.spines[[\"top\", \"right\"]].set_visible(False)\n", + "plt.tight_layout()\n", + "plt.savefig(\"coevo_demo_result.png\", dpi=140, bbox_inches=\"tight\")\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "id": "2ec4019e", + "metadata": {}, + "source": [ + "## \u0412\u044b\u0432\u043e\u0434\n", + "\n", + "\u0418\u0437 \u043e\u0434\u043d\u043e\u0433\u043e \u043f\u0440\u043e\u0441\u0442\u043e\u0433\u043e \u043f\u0440\u043e\u043c\u043f\u0442\u0430 CoEvo \u0441\u043e\u0431\u0440\u0430\u043b \u0441\u0442\u0440\u0443\u043a\u0442\u0443\u0440\u0438\u0440\u043e\u0432\u0430\u043d\u043d\u044b\u0439 \u043f\u0440\u043e\u043c\u043f\u0442 \u0438\u0437 \u0442\u0440\u0451\u0445 \u043f\u043e\u043b\u0435\u0439 \u0438 \u043f\u043e\u0434\u043d\u044f\u043b\n", + "BERTScore. \u0420\u0430\u0437\u0431\u0438\u0435\u043d\u0438\u0435 \u043d\u0430 \u0440\u043e\u043b\u044c / \u0437\u0430\u0434\u0430\u0447\u0443 / \u043e\u0433\u0440\u0430\u043d\u0438\u0447\u0435\u043d\u0438\u044f \u043f\u043e\u0437\u0432\u043e\u043b\u044f\u0435\u0442 \u0443\u043b\u0443\u0447\u0448\u0430\u0442\u044c \u043a\u043e\u043c\u043f\u043e\u043d\u0435\u043d\u0442\u044b \u043d\u0435\u0437\u0430\u0432\u0438\u0441\u0438\u043c\u043e,\n", + "\u0430 \u0440\u0435\u0444\u043b\u0435\u043a\u0441\u0438\u044f \u043d\u0430\u043f\u0440\u0430\u0432\u043b\u044f\u0435\u0442 \u043f\u043e\u0438\u0441\u043a.\n", + "\n", + "\u041f\u043e\u043b\u043d\u043e\u0435 \u0441\u0440\u0430\u0432\u043d\u0435\u043d\u0438\u0435 \u0441 \u0434\u0440\u0443\u0433\u0438\u043c\u0438 \u043c\u0435\u0442\u043e\u0434\u0430\u043c\u0438 \u043d\u0430 \u0448\u0435\u0441\u0442\u0438 \u0434\u0430\u0442\u0430\u0441\u0435\u0442\u0430\u0445 \u2014 \u0432 `benchmark_coevo.png`." + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "CoolPrompt (.venv)", + "language": "python", + "name": "coolprompt-demo" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.12.13" + }, + "widgets": { + "application/vnd.jupyter.widget-state+json": { + "state": { + "031e604e237b4c0a95d74c6b527ed3cb": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "FloatProgressModel", + "state": { + "_dom_classes": [], + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "FloatProgressModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/controls", + "_view_module_version": "2.0.0", + "_view_name": "ProgressView", + "bar_style": "success", + "description": "", + "description_allow_html": false, + "layout": "IPY_MODEL_584f4315bb714773a440f64e9cd909bd", + "max": 103.0, + "min": 0.0, + "orientation": "horizontal", + "style": "IPY_MODEL_fb6ea2a20a9a4b55875f918ae26999f0", + "tabbable": null, + "tooltip": null, + "value": 103.0 + } + }, + "077570ad5c0d48b785739da1415a35b4": { + "model_module": "@jupyter-widgets/base", + "model_module_version": "2.0.0", + "model_name": "LayoutModel", + "state": { + "_model_module": "@jupyter-widgets/base", + "_model_module_version": "2.0.0", + "_model_name": "LayoutModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "LayoutView", + "align_content": null, + "align_items": null, + "align_self": null, + "border_bottom": null, + "border_left": null, + "border_right": null, + "border_top": null, + "bottom": null, + "display": null, + "flex": null, + "flex_flow": null, + "grid_area": null, + "grid_auto_columns": null, + "grid_auto_flow": null, + "grid_auto_rows": null, + "grid_column": null, + "grid_gap": null, + "grid_row": null, + "grid_template_areas": null, + "grid_template_columns": null, + "grid_template_rows": null, + "height": null, + "justify_content": null, + "justify_items": null, + "left": null, + "margin": null, + "max_height": null, + "max_width": null, + "min_height": null, + "min_width": null, + "object_fit": null, + "object_position": null, + "order": null, + "overflow": null, + "padding": null, + "right": null, + "top": null, + "visibility": null, + "width": null + } + }, + "08c8247e7de641859d6d192e53d3c508": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "HBoxModel", + "state": { + "_dom_classes": [], + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "HBoxModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/controls", + "_view_module_version": "2.0.0", + "_view_name": "HBoxView", + "box_style": "", + "children": [ + "IPY_MODEL_2a49562fea2e472bb16c15aae1f8d6c1", + "IPY_MODEL_031e604e237b4c0a95d74c6b527ed3cb", + "IPY_MODEL_82716763d58d4919a89d197bc7a08987" + ], + "layout": "IPY_MODEL_195a14f673404c12ac8918eb5b58f578", + "tabbable": null, + "tooltip": null + } + }, + "195a14f673404c12ac8918eb5b58f578": { + "model_module": "@jupyter-widgets/base", + "model_module_version": "2.0.0", + "model_name": "LayoutModel", + "state": { + "_model_module": "@jupyter-widgets/base", + "_model_module_version": "2.0.0", + "_model_name": "LayoutModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "LayoutView", + "align_content": null, + "align_items": null, + "align_self": null, + "border_bottom": null, + "border_left": null, + "border_right": null, + "border_top": null, + "bottom": null, + "display": null, + "flex": null, + "flex_flow": null, + "grid_area": null, + "grid_auto_columns": null, + "grid_auto_flow": null, + "grid_auto_rows": null, + "grid_column": null, + "grid_gap": null, + "grid_row": null, + "grid_template_areas": null, + "grid_template_columns": null, + "grid_template_rows": null, + "height": null, + "justify_content": null, + "justify_items": null, + "left": null, + "margin": null, + "max_height": null, + "max_width": null, + "min_height": null, + "min_width": null, + "object_fit": null, + "object_position": null, + "order": null, + "overflow": null, + "padding": null, + "right": null, + "top": null, + "visibility": null, + "width": null + } + }, + "203198d1b2f849c38bb3225abfff09ae": { + "model_module": "@jupyter-widgets/base", + "model_module_version": "2.0.0", + "model_name": "LayoutModel", + "state": { + "_model_module": "@jupyter-widgets/base", + "_model_module_version": "2.0.0", + "_model_name": "LayoutModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "LayoutView", + "align_content": null, + "align_items": null, + "align_self": null, + "border_bottom": null, + "border_left": null, + "border_right": null, + "border_top": null, + "bottom": null, + "display": null, + "flex": null, + "flex_flow": null, + "grid_area": null, + "grid_auto_columns": null, + "grid_auto_flow": null, + "grid_auto_rows": null, + "grid_column": null, + "grid_gap": null, + "grid_row": null, + "grid_template_areas": null, + "grid_template_columns": null, + "grid_template_rows": null, + "height": null, + "justify_content": null, + "justify_items": null, + "left": null, + "margin": null, + "max_height": null, + "max_width": null, + "min_height": null, + "min_width": null, + "object_fit": null, + "object_position": null, + "order": null, + "overflow": null, + "padding": null, + "right": null, + "top": null, + "visibility": null, + "width": null + } + }, + "24450aab261649438b0acbaa29232ba6": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "HTMLStyleModel", + "state": { + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "HTMLStyleModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "StyleView", + "background": null, + "description_width": "", + "font_size": null, + "text_color": null + } + }, + "2a49562fea2e472bb16c15aae1f8d6c1": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "HTMLModel", + "state": { + "_dom_classes": [], + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "HTMLModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/controls", + "_view_module_version": "2.0.0", + "_view_name": "HTMLView", + "description": "", + "description_allow_html": false, + "layout": "IPY_MODEL_b5a24bfa925e4def98613ee86b46fad2", + "placeholder": "\u200b", + "style": "IPY_MODEL_5515709d31e74d0786c709788dab537a", + "tabbable": null, + "tooltip": null, + "value": "Loading\u2007weights:\u2007100%" + } + }, + "5515709d31e74d0786c709788dab537a": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "HTMLStyleModel", + "state": { + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "HTMLStyleModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "StyleView", + "background": null, + "description_width": "", + "font_size": null, + "text_color": null + } + }, + "584f4315bb714773a440f64e9cd909bd": { + "model_module": "@jupyter-widgets/base", + "model_module_version": "2.0.0", + "model_name": "LayoutModel", + "state": { + "_model_module": "@jupyter-widgets/base", + "_model_module_version": "2.0.0", + "_model_name": "LayoutModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "LayoutView", + "align_content": null, + "align_items": null, + "align_self": null, + "border_bottom": null, + "border_left": null, + "border_right": null, + "border_top": null, + "bottom": null, + "display": null, + "flex": null, + "flex_flow": null, + "grid_area": null, + "grid_auto_columns": null, + "grid_auto_flow": null, + "grid_auto_rows": null, + "grid_column": null, + "grid_gap": null, + "grid_row": null, + "grid_template_areas": null, + "grid_template_columns": null, + "grid_template_rows": null, + "height": null, + "justify_content": null, + "justify_items": null, + "left": null, + "margin": null, + "max_height": null, + "max_width": null, + "min_height": null, + "min_width": null, + "object_fit": null, + "object_position": null, + "order": null, + "overflow": null, + "padding": null, + "right": null, + "top": null, + "visibility": null, + "width": null + } + }, + "7bbbde5c6f1d4b4a8767656709b2fad5": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "HTMLModel", + "state": { + "_dom_classes": [], + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "HTMLModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/controls", + "_view_module_version": "2.0.0", + "_view_name": "HTMLView", + "description": "", + "description_allow_html": false, + "layout": "IPY_MODEL_f0befdf41bdd4983a71de8cc467dadff", + "placeholder": "\u200b", + "style": "IPY_MODEL_24450aab261649438b0acbaa29232ba6", + "tabbable": null, + "tooltip": null, + "value": "\u2007199/199\u2007[00:00<00:00,\u20076495.81it/s]" + } + }, + "80bd5e6f939342c6b6ad6571ac8e3e20": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "ProgressStyleModel", + "state": { + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "ProgressStyleModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "StyleView", + "bar_color": null, + "description_width": "" + } + }, + "8189b3e8af0b42b4bb9024b06c15e584": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "FloatProgressModel", + "state": { + "_dom_classes": [], + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "FloatProgressModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/controls", + "_view_module_version": "2.0.0", + "_view_name": "ProgressView", + "bar_style": "success", + "description": "", + "description_allow_html": false, + "layout": "IPY_MODEL_077570ad5c0d48b785739da1415a35b4", + "max": 199.0, + "min": 0.0, + "orientation": "horizontal", + "style": "IPY_MODEL_80bd5e6f939342c6b6ad6571ac8e3e20", + "tabbable": null, + "tooltip": null, + "value": 199.0 + } + }, + "82716763d58d4919a89d197bc7a08987": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "HTMLModel", + "state": { + "_dom_classes": [], + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "HTMLModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/controls", + "_view_module_version": "2.0.0", + "_view_name": "HTMLView", + "description": "", + "description_allow_html": false, + "layout": "IPY_MODEL_203198d1b2f849c38bb3225abfff09ae", + "placeholder": "\u200b", + "style": "IPY_MODEL_b25121ecfa3349aebd6d7c2c2a53b64b", + "tabbable": null, + "tooltip": null, + "value": "\u2007103/103\u2007[00:00<00:00,\u20075296.42it/s]" + } + }, + "97ac6d20674a45988b8cbf2033cb72eb": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "HBoxModel", + "state": { + "_dom_classes": [], + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "HBoxModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/controls", + "_view_module_version": "2.0.0", + "_view_name": "HBoxView", + "box_style": "", + "children": [ + "IPY_MODEL_a554639955504d199eeaf5dd024b60e9", + "IPY_MODEL_8189b3e8af0b42b4bb9024b06c15e584", + "IPY_MODEL_7bbbde5c6f1d4b4a8767656709b2fad5" + ], + "layout": "IPY_MODEL_9c19a5097bb14d5da57bf82ffd2562ec", + "tabbable": null, + "tooltip": null + } + }, + "9c19a5097bb14d5da57bf82ffd2562ec": { + "model_module": "@jupyter-widgets/base", + "model_module_version": "2.0.0", + "model_name": "LayoutModel", + "state": { + "_model_module": "@jupyter-widgets/base", + "_model_module_version": "2.0.0", + "_model_name": "LayoutModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "LayoutView", + "align_content": null, + "align_items": null, + "align_self": null, + "border_bottom": null, + "border_left": null, + "border_right": null, + "border_top": null, + "bottom": null, + "display": null, + "flex": null, + "flex_flow": null, + "grid_area": null, + "grid_auto_columns": null, + "grid_auto_flow": null, + "grid_auto_rows": null, + "grid_column": null, + "grid_gap": null, + "grid_row": null, + "grid_template_areas": null, + "grid_template_columns": null, + "grid_template_rows": null, + "height": null, + "justify_content": null, + "justify_items": null, + "left": null, + "margin": null, + "max_height": null, + "max_width": null, + "min_height": null, + "min_width": null, + "object_fit": null, + "object_position": null, + "order": null, + "overflow": null, + "padding": null, + "right": null, + "top": null, + "visibility": null, + "width": null + } + }, + "a554639955504d199eeaf5dd024b60e9": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "HTMLModel", + "state": { + "_dom_classes": [], + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "HTMLModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/controls", + "_view_module_version": "2.0.0", + "_view_name": "HTMLView", + "description": "", + "description_allow_html": false, + "layout": "IPY_MODEL_ab57d75d1029443483be38ee60fec37e", + "placeholder": "\u200b", + "style": "IPY_MODEL_eb7518fe02e74cda9d5b0dfdd010256b", + "tabbable": null, + "tooltip": null, + "value": "Loading\u2007weights:\u2007100%" + } + }, + "ab57d75d1029443483be38ee60fec37e": { + "model_module": "@jupyter-widgets/base", + "model_module_version": "2.0.0", + "model_name": "LayoutModel", + "state": { + "_model_module": "@jupyter-widgets/base", + "_model_module_version": "2.0.0", + "_model_name": "LayoutModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "LayoutView", + "align_content": null, + "align_items": null, + "align_self": null, + "border_bottom": null, + "border_left": null, + "border_right": null, + "border_top": null, + "bottom": null, + "display": null, + "flex": null, + "flex_flow": null, + "grid_area": null, + "grid_auto_columns": null, + "grid_auto_flow": null, + "grid_auto_rows": null, + "grid_column": null, + "grid_gap": null, + "grid_row": null, + "grid_template_areas": null, + "grid_template_columns": null, + "grid_template_rows": null, + "height": null, + "justify_content": null, + "justify_items": null, + "left": null, + "margin": null, + "max_height": null, + "max_width": null, + "min_height": null, + "min_width": null, + "object_fit": null, + "object_position": null, + "order": null, + "overflow": null, + "padding": null, + "right": null, + "top": null, + "visibility": null, + "width": null + } + }, + "b25121ecfa3349aebd6d7c2c2a53b64b": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "HTMLStyleModel", + "state": { + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "HTMLStyleModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "StyleView", + "background": null, + "description_width": "", + "font_size": null, + "text_color": null + } + }, + "b5a24bfa925e4def98613ee86b46fad2": { + "model_module": "@jupyter-widgets/base", + "model_module_version": "2.0.0", + "model_name": "LayoutModel", + "state": { + "_model_module": "@jupyter-widgets/base", + "_model_module_version": "2.0.0", + "_model_name": "LayoutModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "LayoutView", + "align_content": null, + "align_items": null, + "align_self": null, + "border_bottom": null, + "border_left": null, + "border_right": null, + "border_top": null, + "bottom": null, + "display": null, + "flex": null, + "flex_flow": null, + "grid_area": null, + "grid_auto_columns": null, + "grid_auto_flow": null, + "grid_auto_rows": null, + "grid_column": null, + "grid_gap": null, + "grid_row": null, + "grid_template_areas": null, + "grid_template_columns": null, + "grid_template_rows": null, + "height": null, + "justify_content": null, + "justify_items": null, + "left": null, + "margin": null, + "max_height": null, + "max_width": null, + "min_height": null, + "min_width": null, + "object_fit": null, + "object_position": null, + "order": null, + "overflow": null, + "padding": null, + "right": null, + "top": null, + "visibility": null, + "width": null + } + }, + "eb7518fe02e74cda9d5b0dfdd010256b": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "HTMLStyleModel", + "state": { + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "HTMLStyleModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "StyleView", + "background": null, + "description_width": "", + "font_size": null, + "text_color": null + } + }, + "f0befdf41bdd4983a71de8cc467dadff": { + "model_module": "@jupyter-widgets/base", + "model_module_version": "2.0.0", + "model_name": "LayoutModel", + "state": { + "_model_module": "@jupyter-widgets/base", + "_model_module_version": "2.0.0", + "_model_name": "LayoutModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "LayoutView", + "align_content": null, + "align_items": null, + "align_self": null, + "border_bottom": null, + "border_left": null, + "border_right": null, + "border_top": null, + "bottom": null, + "display": null, + "flex": null, + "flex_flow": null, + "grid_area": null, + "grid_auto_columns": null, + "grid_auto_flow": null, + "grid_auto_rows": null, + "grid_column": null, + "grid_gap": null, + "grid_row": null, + "grid_template_areas": null, + "grid_template_columns": null, + "grid_template_rows": null, + "height": null, + "justify_content": null, + "justify_items": null, + "left": null, + "margin": null, + "max_height": null, + "max_width": null, + "min_height": null, + "min_width": null, + "object_fit": null, + "object_position": null, + "order": null, + "overflow": null, + "padding": null, + "right": null, + "top": null, + "visibility": null, + "width": null + } + }, + "fb6ea2a20a9a4b55875f918ae26999f0": { + "model_module": "@jupyter-widgets/controls", + "model_module_version": "2.0.0", + "model_name": "ProgressStyleModel", + "state": { + "_model_module": "@jupyter-widgets/controls", + "_model_module_version": "2.0.0", + "_model_name": "ProgressStyleModel", + "_view_count": null, + "_view_module": "@jupyter-widgets/base", + "_view_module_version": "2.0.0", + "_view_name": "StyleView", + "bar_color": null, + "description_width": "" + } + } + }, + "version_major": 2, + "version_minor": 0 + } + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} \ No newline at end of file diff --git a/notebooks/examples/coevo_demo_result.png b/notebooks/examples/coevo_demo_result.png new file mode 100644 index 00000000..0219f067 Binary files /dev/null and b/notebooks/examples/coevo_demo_result.png differ diff --git a/notebooks/experiments/pipeline/config.example.yaml b/notebooks/experiments/pipeline/config.example.yaml new file mode 100644 index 00000000..7c3ab6bb --- /dev/null +++ b/notebooks/experiments/pipeline/config.example.yaml @@ -0,0 +1,31 @@ +openai_api_keys: + - "sk-..." # key 0 +# - "sk-..." # key 1 (optional, add more for higher RPM) +# active_keys: [0, 1] # indices of keys to use; omit to use all +openrouter_api_key: "" # only needed when provider: openrouter + +provider: openai +seed: 42 +temperature: 0.7 +model: gpt-4o-mini +population_size: 5 +num_epochs: 6 +train_size: 50 +val_size: 80 +use_enhancements: true +use_dedup: true +evolve_constraints: false +factorized_phase_epochs: [4, 4, 3] +output_dir: ../optimization_results +requests_per_minute: + openai: 490 + openrouter: 5000 +datasets_to_run: + - tweet_eval + - xsum + - common_gen + - gsm8k + - squad_v2 + - mediqa +role_modes_to_run: + - coevo_enhanced diff --git a/notebooks/experiments/pipeline/evaluate_prompts.py b/notebooks/experiments/pipeline/evaluate_prompts.py new file mode 100644 index 00000000..69b19126 --- /dev/null +++ b/notebooks/experiments/pipeline/evaluate_prompts.py @@ -0,0 +1,376 @@ +import os +import sys +import json +import yaml +import time +import argparse +import traceback +from datetime import datetime + +_script_dir = os.path.dirname(os.path.abspath(__file__)) +sys.path.append(os.path.abspath(os.path.join(_script_dir, "../../../"))) +sys.path.append(os.path.abspath(os.path.join(_script_dir, "../../../src"))) + +import torch +import gc + +from langchain_core.globals import set_llm_cache +from langchain_community.cache import SQLiteCache + +from coolprompt.optimizer.reflective_prompt.prompt import Prompt +from coolprompt.evaluator import Evaluator, validate_and_create_metric +from coolprompt.utils.logging_config import setup_logging + +from model_utils import create_model, normalize_model_name +from dataset_config import DATASETS_CONFIG, load_eval_data + +setup_logging() + +_EVAL_CONFIG_PATH = os.path.join(_script_dir, "evaluation_config.yaml") +_PROXY_CONFIG_PATH = os.path.join(_script_dir, "proxy_config.yaml") + +with open(_EVAL_CONFIG_PATH) as f: + EVAL_CONFIG = yaml.safe_load(f) + +PROXY_CONFIG = ( + yaml.safe_load(open(_PROXY_CONFIG_PATH)) + if os.path.exists(_PROXY_CONFIG_PATH) + else {} +) + +DEFAULT_TEST_SIZE = EVAL_CONFIG.get("default_test_size", 1000) +DEFAULT_SEED = EVAL_CONFIG.get("default_seed", 42) +DEFAULT_PROVIDER = EVAL_CONFIG.get("provider", "openai") +DEFAULT_MODEL = EVAL_CONFIG.get("model", "gpt-4o-mini") +TEMPERATURE_CONFIG = EVAL_CONFIG.get("temperature", 0.0) + +REQUESTS_PER_MINUTE_CONFIG = EVAL_CONFIG.get( + "requests_per_minute", {"openai": 200, "openrouter": 5000} +) + +_COMBO_MAP = {1: "text_only", 2: "text_role", 3: "text_role_constraints"} +_raw_combo = EVAL_CONFIG.get("combo", "all") +if str(_raw_combo).strip().lower() == "all": + COMBO_FILTER = None +else: + _items = ( + _raw_combo + if isinstance(_raw_combo, list) + else str(_raw_combo).split(",") + ) + COMBO_FILTER = {_COMBO_MAP[int(x)] for x in _items} + +_raw_output_dir = EVAL_CONFIG.get("output_dir", "./evaluation_results") +EVAL_OUTPUT_DIR = ( + _raw_output_dir + if os.path.isabs(_raw_output_dir) + else os.path.join(_script_dir, _raw_output_dir) +) + +_raw_paths = EVAL_CONFIG.get("dataset_paths", {}) +DATASET_PATHS = { + k: ( + [os.path.join(_script_dir, p) if not os.path.isabs(p) else p for p in v] + if isinstance(v, list) + else (os.path.join(_script_dir, v) if not os.path.isabs(v) else v) + ) + for k, v in _raw_paths.items() +} + +_first_proxy = (PROXY_CONFIG.get("proxies") or [None])[0] or PROXY_CONFIG.get( + "proxy", {} +).get("http") +if _first_proxy: + os.environ.setdefault("HTTP_PROXY", _first_proxy) + os.environ.setdefault("HTTPS_PROXY", _first_proxy) + print(f"Proxy set for HF downloads: {_first_proxy}") + + +def evaluate_single_dataset( + dataset_name, + json_path, + inputs, + targets, + provider, + model_name, + requests_per_minute, + seed=None, + full_test=False, +): + if not json_path or not os.path.exists(json_path): + print(f"skip {dataset_name}: no valid json path") + return None + + with open(json_path) as f: + data = json.load(f) + + effective_model_name = normalize_model_name(provider, model_name) + + candidates_meta = data.get("candidates") + if candidates_meta: + prompts_to_eval = [ + { + "combo": c["combo"], + "prompt": c["prompt"], + "role": c.get("role", ""), + "constraints": c.get("constraints", ""), + "val_score": c.get("val_score"), + } + for c in candidates_meta + if c.get("prompt") + ] + else: + best_prompt_text = data.get("best_prompt") + if not best_prompt_text: + print(f"no best_prompt in {dataset_name} json") + return None + prompts_to_eval = [ + { + "combo": "default", + "prompt": best_prompt_text, + "role": data.get("best_role") or "", + "constraints": data.get("best_constraints") or "", + "val_score": data.get("best_score"), + } + ] + + if COMBO_FILTER: + prompts_to_eval = [ + p for p in prompts_to_eval if p["combo"] in COMBO_FILTER + ] + print(f"{dataset_name}: {len(prompts_to_eval)} combos") + + config = DATASETS_CONFIG[dataset_name] + model = create_model( + provider, + model_name, + requests_per_minute, + EVAL_CONFIG, + PROXY_CONFIG, + temperature=TEMPERATURE_CONFIG, + ) + metric = validate_and_create_metric(config["task"], config["metric"]) + evaluator = Evaluator(model, config["task"], metric) + + candidate_results = [] + for p in prompts_to_eval: + prompt_obj = Prompt( + text=p["prompt"], role=p["role"], constraints=p["constraints"] + ) + try: + score = evaluator.evaluate( + prompt=prompt_obj.text, + dataset=inputs, + targets=targets, + system_role=prompt_obj.role or None, + constraints=prompt_obj.constraints or None, + ) + print(f" [{p['combo']}] test={score:.4f} val={p['val_score']}") + candidate_results.append( + { + "combo": p["combo"], + "prompt": p["prompt"], + "role": p["role"], + "constraints": p["constraints"], + "val_score": p["val_score"], + "test_score": score, + } + ) + except Exception as e: + print(f" error [{p['combo']}]: {e}") + traceback.print_exc() + + if not candidate_results: + return None + + best_result = max( + candidate_results, + key=lambda r: ( + r["val_score"] + if r.get("val_score") is not None + else r["test_score"] + ), + ) + print( + f"best: {best_result['combo']} = {best_result['test_score']:.4f} (val={best_result.get('val_score')})" + ) + + opt_params = data.get("parameters", {}) + return { + "dataset": dataset_name, + "role_mode": data.get("role_mode", "unknown"), + "score": best_result["test_score"], + "best_combo": best_result["combo"], + "metric": config["metric"], + "num_samples": len(inputs), + "seed": seed, + "full_test": full_test, + "provider": provider, + "model": effective_model_name, + "opt_temperature": opt_params.get("temperature", 0.7), + "val_temperature": opt_params.get("val_temperature", 0.7), + "use_enhancements": opt_params.get("use_enhancements", None), + "json_path": json_path, + "candidate_results": candidate_results, + "prompt_info": data, + } + + +def main(): + parser = argparse.ArgumentParser(description="Evaluate optimized prompts") + parser.add_argument("--seed", type=int, default=DEFAULT_SEED) + parser.add_argument("--num_samples", type=int, default=DEFAULT_TEST_SIZE) + parser.add_argument("--dataset", type=str) + parser.add_argument( + "--provider", + type=str, + default=DEFAULT_PROVIDER, + choices=["openai", "openrouter"], + ) + parser.add_argument("--model", type=str, default=DEFAULT_MODEL) + parser.add_argument("--requests_per_minute", type=int, default=None) + parser.add_argument( + "--full_test", + action="store_true", + help="Evaluate on the full split (no seed/size limit)", + ) + args = parser.parse_args() + + if args.requests_per_minute is None: + if isinstance(REQUESTS_PER_MINUTE_CONFIG, dict): + args.requests_per_minute = REQUESTS_PER_MINUTE_CONFIG.get( + args.provider, 500 + ) + else: + args.requests_per_minute = int(REQUESTS_PER_MINUTE_CONFIG) + + print( + f"Provider: {args.provider}, Model: {args.model}, RPM: {args.requests_per_minute}" + ) + + set_llm_cache(SQLiteCache(database_path=".langchain.db")) + + output_dir = ( + EVAL_OUTPUT_DIR.rstrip("/\\") + "_full" + if args.full_test + else EVAL_OUTPUT_DIR + ) + + datasets_to_run = DATASET_PATHS + if args.dataset: + if args.dataset not in DATASET_PATHS: + print(f"unknown dataset: {args.dataset}") + return + datasets_to_run = {args.dataset: DATASET_PATHS[args.dataset]} + + results = {} + + for dataset_name, dataset_json_paths in datasets_to_run.items(): + if not dataset_json_paths: + print(f"skip {dataset_name}: no json path") + continue + if dataset_name not in DATASETS_CONFIG: + print(f"skip {dataset_name}: not in config") + continue + + if isinstance(dataset_json_paths, str): + dataset_json_paths = [dataset_json_paths] + elif not isinstance(dataset_json_paths, list): + print(f"skip {dataset_name}: bad path type") + continue + + print(f"\n{dataset_name}") + + config = DATASETS_CONFIG[dataset_name] + try: + inputs, targets = load_eval_data( + dataset_name, + config, + args.num_samples, + args.seed, + args.full_test, + ) + except Exception as e: + print(f"failed to load {dataset_name}: {e}") + continue + + if not inputs: + print(f"no data: {dataset_name}") + continue + + dataset_results = [] + for json_path in dataset_json_paths: + max_retries = 3 + dataset_result = None + run_timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") + + for attempt in range(max_retries): + try: + dataset_result = evaluate_single_dataset( + dataset_name, + json_path, + inputs, + targets, + args.provider, + args.model, + args.requests_per_minute, + seed=args.seed, + full_test=args.full_test, + ) + if dataset_result: + break + except Exception as e: + print(f"error (attempt {attempt + 1}): {e}") + if "429" in str(e) or "Rate limit" in str(e): + time.sleep(60 * (attempt + 1)) + else: + time.sleep(10) + + if dataset_result: + dataset_results.append(dataset_result) + + role_mode = dataset_result.get("role_mode", "unknown") + method_dir = os.path.join(output_dir, dataset_name, role_mode) + os.makedirs(method_dir, exist_ok=True) + score = dataset_result.get("score") + score_str = ( + f"{score:.2f}" if isinstance(score, (int, float)) else "NA" + ) + seed_str = "all" if args.full_test else str(args.seed) + filename = f"{run_timestamp}_{score_str}_{role_mode}_seed{seed_str}.json" + save_path = os.path.join(method_dir, filename) + with open(save_path, "w") as f: + json.dump(dataset_result, f, indent=2) + print(f"saved: {save_path}") + + if dataset_results: + results[dataset_name] = dataset_results + + gc.collect() + torch.cuda.empty_cache() + + print("\nsummary:") + for name, dataset_runs in results.items(): + for idx, res in enumerate(dataset_runs, start=1): + combo_info = ( + f" [{res.get('best_combo', 'default')}]" + if res.get("best_combo") + else "" + ) + print( + f" {name} [run {idx}]{combo_info}: {res['score']:.4f} ({res['metric']}) - {res['num_samples']} samples" + ) + if ( + res.get("candidate_results") + and len(res["candidate_results"]) > 1 + ): + for cr in res["candidate_results"]: + vs = cr.get("val_score") + vs_str = f"{vs:.4f}" if isinstance(vs, float) else str(vs) + print( + f" {cr['combo']}: val={vs_str} test={cr['test_score']:.4f}" + ) + + +if __name__ == "__main__": + main() diff --git a/notebooks/experiments/pipeline/evaluation_config.example.yaml b/notebooks/experiments/pipeline/evaluation_config.example.yaml new file mode 100644 index 00000000..9ee6ded4 --- /dev/null +++ b/notebooks/experiments/pipeline/evaluation_config.example.yaml @@ -0,0 +1,18 @@ +openai_api_keys: + - "sk-..." # key 0 +# active_keys: [0] # indices of keys to use; omit to use all +openrouter_api_key: "" # only needed when provider: openrouter + +provider: openai +model: gpt-4o-mini +temperature: 0.0 +default_test_size: 1000 +default_seed: 42 +requests_per_minute: + openai: 200 + openrouter: 5000 +combo: all # all | 1 (text_only) | 2 (text_role) | 3 (text_role_constraints) +output_dir: ./evaluation_results + +# Paths to optimization result JSON files, populated automatically by optmize_single.py +dataset_paths: {} diff --git a/notebooks/experiments/pipeline/model_utils.py b/notebooks/experiments/pipeline/model_utils.py new file mode 100644 index 00000000..b0dcd9df --- /dev/null +++ b/notebooks/experiments/pipeline/model_utils.py @@ -0,0 +1,129 @@ +import os +import time +import httpx +from langchain_openai import ChatOpenAI +from langchain_core.rate_limiters import InMemoryRateLimiter + + +class MultiKeyModel: + _MAX_INNER_CONCURRENCY = 8 + + def __init__(self, models: list): + self._models = models + + def batch(self, requests: list) -> list: + n = len(self._models) + if n == 1: + return self._batch_with_retry(self._models[0], requests) + + chunks = [[] for _ in range(n)] + chunk_indices = [[] for _ in range(n)] + for i, req in enumerate(requests): + slot = i % n + chunks[slot].append(req) + chunk_indices[slot].append(i) + + results = [None] * len(requests) + for i in range(n): + if not chunks[i]: + continue + responses = self._batch_with_retry(self._models[i], chunks[i]) + for idx, response in zip(chunk_indices[i], responses): + results[idx] = response + return results + + def _batch_with_retry(self, model, requests: list, max_attempts: int = 4) -> list: + c = self._MAX_INNER_CONCURRENCY + for attempt in range(max_attempts): + try: + return model.batch(requests, config={"max_concurrency": c}) + except Exception: + if attempt < max_attempts - 1: + c = max(1, c // 2) + time.sleep(5 * (attempt + 1)) + else: + raise + + def invoke(self, request): + return self._models[0].invoke(request) + + def __getattr__(self, name): + return getattr(self._models[0], name) + + +def load_proxy_list(proxy_config: dict) -> list: + proxy_list = proxy_config.get("proxies") or [] + if not proxy_list: + legacy = proxy_config.get("proxy", {}).get("http") + if legacy: + proxy_list = [legacy] + return proxy_list + + +def normalize_model_name(provider: str, model_name: str) -> str: + if provider == "openrouter" and "/" not in model_name: + return f"openai/{model_name}" + if provider == "openai" and model_name.startswith("openai/"): + return model_name.split("/", 1)[1] + return model_name + + +def _resolve_api_keys(config: dict) -> list: + all_keys = config.get("openai_api_keys") or ( + [config["openai_api_key"]] if config.get("openai_api_key") else [os.getenv("OPENAI_API_KEY")] + ) + active = config.get("active_keys") + keys = [all_keys[i] for i in active if i < len(all_keys)] if active is not None else all_keys + return [k for k in keys if k] + + +def _build_models(api_keys, model_name, base_url, temperature, proxy_list, model_kwargs, requests_per_minute=None): + models = [] + for idx, key in enumerate(api_keys): + key_http_client = None + if proxy_list: + proxy_url = proxy_list[idx % len(proxy_list)] + key_http_client = httpx.Client(proxy=proxy_url, timeout=60.0) + rate_limiter = None + if requests_per_minute: + rate_limiter = InMemoryRateLimiter( + requests_per_second=requests_per_minute / 60.0, + check_every_n_seconds=0.1, + max_bucket_size=requests_per_minute, + ) + models.append(ChatOpenAI( + model=model_name, + api_key=key, + base_url=base_url, + temperature=temperature, + rate_limiter=rate_limiter, + max_retries=3, + model_kwargs=model_kwargs, + http_client=key_http_client, + )) + return MultiKeyModel(models) if len(models) > 1 else models[0] + + +def create_model(provider, model_name, requests_per_minute, config, proxy_config, temperature=0.0): + proxy_list = load_proxy_list(proxy_config) + model_name = normalize_model_name(provider, model_name) + + if provider == "openrouter": + api_keys = [config.get("openrouter_api_key") or os.getenv("OPENROUTER_API_KEY")] + base_url = "https://openrouter.ai/api/v1" + model_kwargs = {"extra_body": {"provider": {"order": ["openai"], "allow_fallbacks": False}}} + else: + api_keys = _resolve_api_keys(config) + base_url = None + model_kwargs = {} + + api_keys = [k for k in api_keys if k] + if not api_keys: + raise ValueError("No API keys found in config") + + if requests_per_minute: + print(f"{len(api_keys)} keys, {requests_per_minute} RPM (total {len(api_keys) * requests_per_minute})") + if proxy_list: + print(f"{len(proxy_list)} proxies") + + return _build_models(api_keys, model_name, base_url, temperature, proxy_list, model_kwargs, requests_per_minute) diff --git a/notebooks/experiments/pipeline/optmize_single.py b/notebooks/experiments/pipeline/optmize_single.py new file mode 100644 index 00000000..7d148e1b --- /dev/null +++ b/notebooks/experiments/pipeline/optmize_single.py @@ -0,0 +1,452 @@ +import os +import sys +import json +import yaml +import time +import random +import argparse +import traceback +from datetime import datetime + +_script_dir = os.path.dirname(os.path.abspath(__file__)) +sys.path.append(os.path.abspath(os.path.join(_script_dir, "../../../"))) +sys.path.append(os.path.abspath(os.path.join(_script_dir, "../../../src"))) + +import numpy as np +import torch +import gc + +from coolprompt.optimizer.reflective_prompt.evoluter import ReflectiveEvoluter +from coolprompt.optimizer.reflective_prompt.factorized_evoluter import ( + FactorizedEvoluter, +) +from coolprompt.optimizer.reflective_prompt.coevo_evoluter import ( + CoevoEvoluter, + PerFieldCoevoEvoluter, +) +from coolprompt.evaluator import Evaluator, validate_and_create_metric +from coolprompt.utils.logging_config import setup_logging + +from model_utils import create_model +from dataset_config import DATASETS_CONFIG, load_train_data + +setup_logging() + +_EVAL_CONFIG_PATH = os.path.join(_script_dir, "evaluation_config.yaml") +_CONFIG_PATH = os.path.join(_script_dir, "config.yaml") +_PROXY_CONFIG_PATH = os.path.join(_script_dir, "proxy_config.yaml") + +with open(_CONFIG_PATH) as f: + CONFIG = yaml.safe_load(f) + +PROXY_CONFIG = ( + yaml.safe_load(open(_PROXY_CONFIG_PATH)) + if os.path.exists(_PROXY_CONFIG_PATH) + else {} +) + +if CONFIG.get("openai_api_key"): + os.environ["OPENAI_API_KEY"] = CONFIG["openai_api_key"] +if CONFIG.get("openrouter_api_key"): + os.environ["OPENROUTER_API_KEY"] = CONFIG["openrouter_api_key"] + +_FACTORIZED_MODES = {"factorized", "factorized_dedup", "factorized_top_prompts"} +_VALID_ROLE_MODES = { + "with_role", + "no_role", + "coevo", + "coevo_enhanced", + "coevo_no_enhancements", + "coevo_per_field", +} | _FACTORIZED_MODES + + +def _update_eval_config(dataset_name: str, result_file: str) -> None: + with open(_EVAL_CONFIG_PATH) as f: + eval_cfg = yaml.safe_load(f) + rel_path = os.path.relpath(result_file, os.path.dirname(_EVAL_CONFIG_PATH)) + if os.sep != "/": + rel_path = rel_path.replace(os.sep, "/") + if "dataset_paths" not in eval_cfg or eval_cfg["dataset_paths"] is None: + eval_cfg["dataset_paths"] = {} + eval_cfg["dataset_paths"][dataset_name] = [rel_path] + with open(_EVAL_CONFIG_PATH, "w") as f: + yaml.dump(eval_cfg, f, allow_unicode=True, sort_keys=False) + print(f"evaluation_config.yaml updated: {dataset_name} -> {rel_path}") + + +def run_optimization( + args, + config, + train_inputs, + train_targets, + val_inputs, + val_targets, + logs_dir, + role_mode, + settings, +): + temperature = settings["temperature"] + model = create_model( + args.provider, + args.model, + args.requests_per_minute, + CONFIG, + PROXY_CONFIG, + temperature=temperature, + ) + val_model = create_model( + args.provider, args.model, None, CONFIG, PROXY_CONFIG, temperature=0.0 + ) + + metric = validate_and_create_metric(config["task"], config["metric"]) + evaluator = Evaluator(model, config["task"], metric) + val_evaluator = Evaluator(val_model, config["task"], metric) + + pop_size = settings["population_size"] + num_epochs = settings["num_epochs"] + phase_epochs = settings["factorized_phase_epochs"] + use_enhancements = settings["use_enhancements"] + use_dedup = settings["use_dedup"] + evolve_constraints = settings["evolve_constraints"] + task_desc = config["initial_task_description"] + initial_constraints = config.get("initial_output_constraints", "") + + print(f"\nrunning: {role_mode}") + + if role_mode in _FACTORIZED_MODES: + evoluter = FactorizedEvoluter( + model=model, + evaluator=evaluator, + train_dataset=train_inputs, + train_targets=train_targets, + validation_dataset=val_inputs, + validation_targets=val_targets, + problem_description=f"Task: {config['description']}", + initial_prompt=task_desc, + initial_role=config["initial_system_behavior"], + initial_constraints=( + initial_constraints if evolve_constraints else None + ), + population_size=pop_size, + phase_epochs=phase_epochs, + run_constraints_phase=evolve_constraints, + use_cache=True, + output_path=logs_dir, + use_enhancements=use_enhancements, + use_dedup=use_dedup, + val_evaluator=val_evaluator, + ) + elif role_mode in ("coevo_enhanced", "coevo_no_enhancements"): + evoluter = CoevoEvoluter( + model=model, + evaluator=evaluator, + train_dataset=train_inputs, + train_targets=train_targets, + validation_dataset=val_inputs, + validation_targets=val_targets, + problem_description=f"Task: {config['description']}", + initial_prompt=task_desc, + initial_role=config["initial_system_behavior"], + initial_constraints=initial_constraints, + population_size=pop_size, + num_epochs=num_epochs, + use_cache=True, + output_path=logs_dir, + use_enhancements=(role_mode == "coevo_enhanced"), + val_evaluator=val_evaluator, + ) + elif role_mode == "coevo_per_field": + evoluter = PerFieldCoevoEvoluter( + model=model, + evaluator=evaluator, + train_dataset=train_inputs, + train_targets=train_targets, + validation_dataset=val_inputs, + validation_targets=val_targets, + problem_description=f"Task: {config['description']}", + initial_prompt=task_desc, + initial_role=config["initial_system_behavior"], + initial_constraints=initial_constraints, + population_size=pop_size, + num_epochs=num_epochs, + use_cache=True, + output_path=logs_dir, + use_enhancements=True, + val_evaluator=val_evaluator, + ) + else: + if role_mode == "coevo": + initial_role, evolve_role = config["initial_system_behavior"], True + elif role_mode == "with_role": + initial_role, evolve_role = config["initial_system_behavior"], False + else: + initial_role, evolve_role = "", False + evoluter = ReflectiveEvoluter( + model=model, + evaluator=evaluator, + train_dataset=train_inputs, + train_targets=train_targets, + validation_dataset=val_inputs, + validation_targets=val_targets, + problem_description=f"Task: {config['description']}", + initial_prompt=task_desc, + initial_role=initial_role, + initial_constraints=( + initial_constraints if evolve_constraints else None + ), + evolve_role=evolve_role, + evolve_constraints=evolve_constraints, + population_size=pop_size, + num_epochs=num_epochs, + use_cache=True, + output_path=logs_dir, + use_enhancements=use_enhancements, + val_evaluator=val_evaluator, + ) + + evoluter.evolution() + return evoluter + + +def main(): + parser = argparse.ArgumentParser( + description="Optimize prompt for multiple datasets" + ) + parser.add_argument( + "--provider", + type=str, + default=CONFIG["provider"], + choices=["openai", "openrouter"], + ) + parser.add_argument("--model", type=str, default=CONFIG["model"]) + _output_dir = CONFIG["output_dir"] + if not os.path.isabs(_output_dir): + _output_dir = os.path.abspath(os.path.join(_script_dir, _output_dir)) + parser.add_argument("--output_dir", type=str, default=_output_dir) + parser.add_argument("--requests_per_minute", type=int, default=None) + parser.add_argument( + "--debug", + action="store_true", + help="Minimal sizes (train=5, val=5, pop=2, epochs=1)", + ) + args = parser.parse_args() + + if args.requests_per_minute is None: + args.requests_per_minute = CONFIG["requests_per_minute"].get( + args.provider, 500 + ) + + settings = { + "population_size": CONFIG["population_size"], + "num_epochs": CONFIG["num_epochs"], + "train_size": CONFIG["train_size"], + "val_size": CONFIG["val_size"], + "factorized_phase_epochs": tuple( + CONFIG.get("factorized_phase_epochs", [4, 3, 3]) + ), + "temperature": CONFIG.get("temperature", 0.0), + "use_enhancements": CONFIG.get("use_enhancements", True), + "use_dedup": CONFIG.get("use_dedup", True), + "evolve_constraints": CONFIG.get("evolve_constraints", False), + } + + if args.debug: + settings.update( + { + "population_size": 2, + "num_epochs": 1, + "train_size": 5, + "val_size": 5, + "factorized_phase_epochs": (1, 1, 1), + } + ) + print("Debug mode: train=5, val=5, pop=2, epochs=1") + + _seed = CONFIG.get("seed", 42) + random.seed(_seed) + np.random.seed(_seed) + + datasets_to_run = CONFIG.get("datasets_to_run", []) + role_modes_to_run = CONFIG.get("role_modes_to_run", ["with_role"]) + + print(f"Datasets: {datasets_to_run}") + print(f"Role modes: {role_modes_to_run}") + print( + f"Provider: {args.provider}, Model: {args.model}, RPM: {args.requests_per_minute}" + ) + print(f"Enhancements: {'ON' if settings['use_enhancements'] else 'OFF'}") + print( + f"Constraints evolution: {'ON' if settings['evolve_constraints'] else 'OFF'}" + ) + + for dataset_name in datasets_to_run: + if dataset_name not in DATASETS_CONFIG: + print(f"skip unknown dataset: {dataset_name}") + continue + + print(f"\n{dataset_name}") + + config = DATASETS_CONFIG[dataset_name] + run_dir = os.path.join(args.output_dir, dataset_name) + os.makedirs(run_dir, exist_ok=True) + + train_size = settings["train_size"] + val_size = settings["val_size"] + try: + inputs, targets = load_train_data( + dataset_name, config, num_samples=train_size + val_size + 20 + ) + except Exception as e: + print(f"failed to load {dataset_name}: {e}") + continue + + if len(inputs) < train_size + val_size: + print( + f"warning: only {len(inputs)} samples, need {train_size + val_size}" + ) + + train_inputs = inputs[:train_size] + train_targets = targets[:train_size] + val_inputs = inputs[train_size : train_size + val_size] + val_targets = targets[train_size : train_size + val_size] + print(f"Split: train={len(train_inputs)}, val={len(val_inputs)}") + + for role_mode in role_modes_to_run: + if role_mode not in _VALID_ROLE_MODES: + print(f"skip unknown mode: {role_mode}") + continue + + method_dir = os.path.join(run_dir, role_mode) + os.makedirs(method_dir, exist_ok=True) + + timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") + logs_dir = os.path.join( + method_dir, "logs", f"logs_{role_mode}_{timestamp}" + ) + os.makedirs(logs_dir, exist_ok=True) + print(f"\nmode: {role_mode}") + print(f"Logs: {logs_dir}") + + max_retries = 3 + evoluter = None + start_time = time.time() + + for attempt in range(max_retries): + try: + if attempt > 0: + print(f"Retry {attempt + 1}/{max_retries}...") + time.sleep(60) + evoluter = run_optimization( + args, + config, + train_inputs, + train_targets, + val_inputs, + val_targets, + logs_dir, + role_mode, + settings, + ) + break + except Exception as e: + error_msg = str(e) + print( + f"error {dataset_name}/{role_mode} (attempt {attempt + 1}): {error_msg}" + ) + if ( + "429" in error_msg + or "Rate limit" in error_msg + or "quota" in error_msg + ): + wait_time = (attempt + 1) * 60 + print(f"Rate limit. Waiting {wait_time}s...") + time.sleep(wait_time) + else: + time.sleep(30) + if attempt == max_retries - 1: + print(f"max retries: {dataset_name}/{role_mode}") + with open( + os.path.join( + method_dir, + f"error_log_{role_mode}_{timestamp}.txt", + ), + "w", + ) as f: + f.write( + f"Failed after {max_retries} attempts.\nLast error: {error_msg}\n{traceback.format_exc()}" + ) + + duration = time.time() - start_time + + if evoluter: + is_factorized = role_mode in _FACTORIZED_MODES + result_data = { + "dataset": dataset_name, + "role_mode": role_mode, + "model": args.model, + "best_prompt": evoluter.best_prompt_overall, + "best_role": evoluter.best_role_overall, + "best_constraints": evoluter.best_constraints_overall or "", + "best_score": evoluter.best_score_overall, + "candidates": getattr(evoluter, "candidates", None), + "initial_task_description": evoluter.initial_prompt, + "initial_system_behavior": evoluter.initial_role or "", + "initial_output_constraints": evoluter.initial_constraints + or "", + "description": config["description"], + "parameters": { + "population_size": settings["population_size"], + "num_epochs": ( + None if is_factorized else settings["num_epochs"] + ), + "factorized_phase_epochs": ( + list(settings["factorized_phase_epochs"]) + if is_factorized + else None + ), + "train_size": len(train_inputs), + "val_size": len(val_inputs), + "rate_limit_rpm": args.requests_per_minute, + "provider": args.provider, + "temperature": settings["temperature"], + "val_temperature": 0.0, + "use_enhancements": ( + (role_mode == "coevo_enhanced") + if role_mode + in ("coevo_enhanced", "coevo_no_enhancements") + else settings["use_enhancements"] + ), + "evolve_constraints": settings["evolve_constraints"], + }, + "duration_seconds": duration, + "timestamp": timestamp, + } + + score = evoluter.best_score_overall + score_str = ( + f"{score:.2f}" if isinstance(score, (int, float)) else "NA" + ) + result_filename = ( + f"{timestamp}_{score_str}_{role_mode}_seed{_seed}.json" + ) + result_file = os.path.join(method_dir, result_filename) + with open(result_file, "w") as f: + json.dump(result_data, f, indent=2) + + _update_eval_config(dataset_name, result_file) + + print( + f"\ndone {dataset_name}/{role_mode}, score={evoluter.best_score_overall}" + ) + print(f"saved: {result_file}") + + del evoluter + gc.collect() + torch.cuda.empty_cache() + + print("\ndone") + + +if __name__ == "__main__": + main() diff --git a/requirements.txt b/requirements.txt index d18ee6fb..2c9251d7 100644 --- a/requirements.txt +++ b/requirements.txt @@ -15,4 +15,5 @@ langchain_huggingface>=0.3.1 langchain-openai>=0.3.30 langdetect>=0.4.31 deepeval>=3.7.2 -transformers<5.0.0 \ No newline at end of file +transformers<5.0.0 +sentence-transformers>=3.0.0 \ No newline at end of file