diff --git a/workflows/README.md b/workflows/README.md index a51d7ea8..f7a440b7 100644 --- a/workflows/README.md +++ b/workflows/README.md @@ -14,9 +14,9 @@ Generate workflows **without LLM for step creation** - 10-100x faster, guarantee ### 2. Create Your Own Workflow ```python from workflow_use.healing.service import HealingService -from langchain_anthropic import ChatAnthropic +from browser_use.llm import ChatBrowserUse -llm = ChatAnthropic(model_name="claude-3-5-sonnet-20241022") +llm = ChatBrowserUse(model_name="bu-latest") service = HealingService(llm=llm, use_deterministic_conversion=True) workflow = await service.generate_workflow_from_prompt( diff --git a/workflows/docs/DETERMINISTIC.md b/workflows/docs/DETERMINISTIC.md index 2571a634..cd1e7cb2 100644 --- a/workflows/docs/DETERMINISTIC.md +++ b/workflows/docs/DETERMINISTIC.md @@ -7,10 +7,10 @@ Generate workflows **without LLM for step creation** - directly map browser acti ```python import asyncio from workflow_use.healing.service import HealingService -from langchain_anthropic import ChatAnthropic +from browser_use.llm import ChatBrowserUse async def main(): - llm = ChatAnthropic(model_name="claude-3-5-sonnet-20241022") + llm = ChatBrowserUse(model_name="bu-latest") service = HealingService(llm=llm, use_deterministic_conversion=True) diff --git a/workflows/workflow_use/healing/deterministic_converter.py b/workflows/workflow_use/healing/deterministic_converter.py index 0a2e5ee6..714b974f 100644 --- a/workflows/workflow_use/healing/deterministic_converter.py +++ b/workflows/workflow_use/healing/deterministic_converter.py @@ -19,6 +19,8 @@ class DeterministicWorkflowConverter: def __init__(self): self.element_text_map: Dict[str, str] = {} # Maps element hashes to visible text + self.element_hash_map: Dict[int, str] = {} # Maps element index to hash for selector population + self.captured_element_text_map: Dict[int, Any] = {} # Captured during agent execution def convert_history_to_steps(self, history_list: AgentHistoryList) -> List[Dict[str, Any]]: """ @@ -36,46 +38,263 @@ def convert_history_to_steps(self, history_list: AgentHistoryList) -> List[Dict[ if history.model_output is None: continue + # Capture semantic context from the agent's reasoning + # current_state is an AgentBrain object, extract the text from it + current_state = getattr(history.model_output, 'current_state', None) + reasoning_text = None + if current_state: + # AgentBrain has various fields, extract the most relevant one + # Try memory first, then thought, then convert to string + if hasattr(current_state, 'memory') and current_state.memory: + reasoning_text = str(current_state.memory) + elif hasattr(current_state, 'thought') and current_state.thought: + reasoning_text = str(current_state.thought) + elif hasattr(current_state, 'evaluation_previous_goal') and current_state.evaluation_previous_goal: + reasoning_text = str(current_state.evaluation_previous_goal) + else: + reasoning_text = str(current_state) + + agent_context = { + 'reasoning': reasoning_text, + 'page_url': getattr(history.state, 'url', None), + 'page_title': getattr(history.state, 'title', None), + } + # Process each action in this history item for action in history.model_output.action: action_dict = action.model_dump() - action_type = action_dict.get('type', '') + + # Browser-use action format: {action_type: {params}} + # Extract action type from dictionary keys (excluding 'type' if present) + action_type = None + action_params = {} + + for key, value in action_dict.items(): + if key != 'type' and isinstance(value, dict): + action_type = key + action_params = value + break + + if not action_type: + # Fallback to old format if present + action_type = action_dict.get('type', '') + action_params = action_dict + + # Debug: Log the action type and params + print(f'šŸ” Processing action type: "{action_type}"') + print(f' Action params: {action_params}') + reasoning = agent_context.get('reasoning') + if reasoning: + # Truncate long reasoning text + reasoning_preview = reasoning[:150] + '...' if len(reasoning) > 150 else reasoning + print(f' 🧠 Agent reasoning: {reasoning_preview}') # Get interacted element data if available - element_data = self._get_element_data(history, action_dict) + element_data = self._get_element_data(history, action_params) - # Convert action to semantic step - step = self._convert_action_to_step(action_type, action_dict, element_data) + # Convert action to semantic step with context + step = self._convert_action_to_step(action_type, action_params, element_data, agent_context) if step: + print(f' āœ… Converted to step: {step.get("type")}') steps.append(step) + else: + print(' āŒ Skipped (no step generated)') + print(f'\nšŸ“Š Total steps generated: {len(steps)}') return steps def _get_element_data(self, history, action_dict: Dict[str, Any]) -> Optional[Dict[str, Any]]: """ - Extract element data from interacted elements using the action's index. + Extract element data from the DOM using the box overlay index. - Returns element data including visible text, attributes, node_name, etc. + Browser-use creates overlay boxes on top of elements. The index refers to the box, + not the actual element. We need to look at the DOM state to find the visible text. """ index = action_dict.get('index') if index is None: return None - # Browser-use uses 1-based indexing - element_index = index - 1 if index > 0 else 0 - + # Debug: Print available data in history.state + print(f' šŸ” Looking for element with index {index}') + + # First, try to get from captured element map (captured during agent execution) + if index in self.captured_element_text_map: + element_info = self.captured_element_text_map[index] + print(f' āœ… Found element {index} in captured map!') + print(f' Element info: {element_info}') + # Normalize and return + normalized = self._normalize_element_data(element_info) + if normalized: + self.element_hash_map[index] = normalized['element_hash'] + return normalized + + # Print the full state dict to see what browser-use provides + try: + state_dict = history.state.to_dict() + print(f' State dict keys: {state_dict.keys()}') + + # Check if tabs have element information + if 'tabs' in state_dict and state_dict['tabs']: + first_tab = state_dict['tabs'][0] + print(f' First tab keys: {first_tab.keys()}') + + # Look for selector map or interactive elements + if 'selector_map' in first_tab: + print(f' Found selector_map with {len(first_tab["selector_map"])} entries') + # Check if our index is in the selector map + if str(index) in first_tab['selector_map']: + element_info = first_tab['selector_map'][str(index)] + print(f' āœ… Found element {index} in selector_map!') + print(f' Element info: {element_info}') + # Normalize and return + normalized = self._normalize_element_data(element_info) + if normalized: + self.element_hash_map[index] = normalized['element_hash'] + return normalized + + # Check for interactive_elements field + if 'interactive_elements' in first_tab: + elements = first_tab['interactive_elements'] + print(f' Found interactive_elements with {len(elements)} entries') + # Find element by index + for elem in elements: + if elem.get('index') == index or elem.get('highlight_index') == index: + print(f' āœ… Found element {index} in interactive_elements!') + print(f' Element: {elem}') + # Normalize and return + normalized = self._normalize_element_data(elem) + if normalized: + self.element_hash_map[index] = normalized['element_hash'] + return normalized + except Exception as e: + print(f' Error accessing state dict: {e}') + + # Fallback to old method interacted_elements = history.state.interacted_element - if element_index < len(interacted_elements): - element = interacted_elements[element_index] - if element is None: - return None + print(f' Number of interacted elements: {len(interacted_elements)}') + + # Try to find by highlight_index (the box number) + matching_element = None + for i, element in enumerate(interacted_elements): + if element: + if hasattr(element, 'highlight_index') and element.highlight_index == index: + matching_element = element + print(' āœ“ Found by highlight_index match') + break + + if matching_element is None: + print(f' āš ļø Could not find element with index {index} - returning None') + return None + + # Normalize the element data + normalized = self._normalize_element_data(matching_element) + if normalized: + self.element_hash_map[index] = normalized['element_hash'] + print( + f' šŸ“ Found element: tag={normalized.get("node_name")}, ' + f'value="{normalized.get("node_value")[:50] if normalized.get("node_value") else ""}"' + ) + print(f' Attributes: {list(normalized.get("attributes", {}).keys())}') + print(f' Hash: {normalized["element_hash"]}') + + return normalized + + def _create_semantic_description( + self, action_type: str, base_description: str, agent_context: Dict[str, Any], target_text: Optional[str] = None + ) -> str: + """ + Create a semantically rich description using agent reasoning and context. + + Args: + action_type: The type of action + base_description: The basic description + agent_context: Context from agent reasoning + target_text: Optional target text for the action + + Returns: + Enhanced description with semantic context + """ + reasoning = agent_context.get('reasoning') or '' + page_title = agent_context.get('page_title') or '' + + # If we have agent reasoning, try to extract intent + if reasoning and isinstance(reasoning, str): + # Simple heuristic: extract action intent from reasoning + reasoning_lower = reasoning.lower() + + # Look for intent keywords + intent_map = { + 'click': ['click', 'select', 'choose', 'open'], + 'navigation': ['navigate', 'go to', 'visit', 'open'], + 'input': ['enter', 'type', 'input', 'fill'], + 'scroll': ['scroll', 'view more', 'see more'], + 'extract': ['extract', 'get', 'find', 'collect'], + } + + # Find matching intent + for intent_type, keywords in intent_map.items(): + if action_type in ['click', 'click_element'] and intent_type == 'click': + for keyword in keywords: + if keyword in reasoning_lower and target_text: + # Try to find what they're clicking on + if 'section' in reasoning_lower: + return f"Click on '{target_text}' to access section" + elif 'filing' in reasoning_lower or 'sec' in reasoning_lower: + return f"Click on '{target_text}' (Filings section)" + elif 'news' in reasoning_lower or 'press' in reasoning_lower: + return f"Click on '{target_text}' (News/Press Releases)" + elif 'event' in reasoning_lower or 'webcast' in reasoning_lower: + return f"Click on '{target_text}' (Events/Webcasts)" + elif 'presentation' in reasoning_lower: + return f"Click on '{target_text}' (Presentations)" + + # Fallback to base description with page context + if page_title and action_type in ['click', 'input']: + return f'{base_description} (on {page_title})' + + return base_description + + def _normalize_element_data(self, raw_data: Any) -> Dict[str, Any]: + """ + Normalize element data from various browser-use formats to a consistent structure. + """ + import hashlib + + # If it's already a dict from selector_map or interactive_elements + if isinstance(raw_data, dict): + # Extract common fields with fallbacks + result = { + 'node_name': raw_data.get('tag_name') or raw_data.get('node_name') or '', + 'node_value': raw_data.get('text') or raw_data.get('node_value') or '', + 'attributes': raw_data.get('attributes', {}), + 'xpath': raw_data.get('xpath') or raw_data.get('x_path') or '', + } + + # Compute element hash if we have the data + tag_name = result['node_name'].lower() + # Use xpath or a combination of attributes as hash source + hash_source = result['xpath'] or str(result['attributes']) + element_hash = hashlib.sha256(f'{tag_name}_{hash_source}'.encode()).hexdigest()[:10] + + result['element_hash'] = element_hash + result['element_object'] = raw_data # Store raw for reference + + return result + + # If it's a DOM element object (fallback) + if hasattr(raw_data, 'node_name'): + tag_name = raw_data.node_name.lower() if hasattr(raw_data, 'node_name') else '' + element_browser_hash = getattr(raw_data, 'element_hash', '') + element_hash = hashlib.sha256(f'{tag_name}_{element_browser_hash}'.encode()).hexdigest()[:10] return { - 'node_name': getattr(element, 'node_name', ''), - 'node_value': getattr(element, 'node_value', ''), - 'attributes': getattr(element, 'attributes', {}), - 'xpath': getattr(element, 'x_path', ''), + 'node_name': getattr(raw_data, 'node_name', ''), + 'node_value': getattr(raw_data, 'node_value', ''), + 'attributes': getattr(raw_data, 'attributes', {}), + 'xpath': getattr(raw_data, 'x_path', ''), + 'element_hash': element_hash, + 'element_object': raw_data, } return None @@ -86,77 +305,170 @@ def _extract_target_text(self, element_data: Optional[Dict[str, Any]], action_di Priority: 1. Visible text content (node_value) - 2. Placeholder attribute - 3. aria-label attribute - 4. title attribute - 5. Value attribute - 6. Input text being entered (for input actions) + 2. aria-label attribute + 3. title attribute + 4. placeholder attribute + 5. alt attribute (for images) + 6. value attribute + 7. name attribute + 8. id attribute (last resort) + 9. href attribute (for anchor tags) - extract meaningful part + 10. Input text being entered (for input actions) + 11. Node name + xpath hint (absolute fallback) """ if not element_data: # For input actions, use the text being entered as fallback if action_dict.get('text'): return action_dict['text'] - return '' + return 'element' # Priority 1: Visible text content node_value = element_data.get('node_value', '').strip() if node_value: + print(f' āœ“ Using node_value as target_text: "{node_value}"') return node_value - # Priority 2-5: Check attributes + # Priority 2-8: Check attributes in order attributes = element_data.get('attributes', {}) - for attr in ['placeholder', 'aria-label', 'title', 'value']: + for attr in ['aria-label', 'title', 'placeholder', 'alt', 'value', 'name', 'id']: if attr in attributes and attributes[attr]: - return attributes[attr] - - # For input actions, use the text being entered + text = str(attributes[attr]).strip() + if text: + print(f' āœ“ Using {attr} attribute as target_text: "{text}"') + return text + + # Priority 9: For anchor tags, extract meaningful text from href + node_name = element_data.get('node_name', '') + if node_name == 'a' and 'href' in attributes: + href = attributes['href'] + if isinstance(href, str): + # Remove query params and anchors + href = href.split('?')[0].split('#')[0] + # Get the last path segment + path_parts = href.rstrip('/').split('/') + if path_parts: + last_part = path_parts[-1] + # Convert URL-friendly text to readable text + # E.g., "sec-filings" -> "SEC Filings" + # Skip generic terms + skip_terms = ['www.edison.com', 'edison.com', 'investors', 'www', 'com', 'http:', 'https:'] + if last_part and last_part not in skip_terms: + text = last_part.replace('-', ' ').replace('_', ' ').title() + print(f' āœ“ Extracted from href as target_text: "{text}"') + return text + + # Priority 10: For input actions, use the text being entered if action_dict.get('text'): - return action_dict['text'] + text = action_dict['text'] + print(f' āœ“ Using input text as target_text: "{text}"') + return text + + # Priority 11: Fallback - use node name as hint + if node_name: + print(f' āš ļø No good target text found, using node name: "{node_name}"') + return f'{node_name} element' - return '' + print(' āš ļø No target text found at all') + return 'element' def _convert_action_to_step( - self, action_type: str, action_dict: Dict[str, Any], element_data: Optional[Dict[str, Any]] + self, + action_type: str, + action_dict: Dict[str, Any], + element_data: Optional[Dict[str, Any]], + agent_context: Optional[Dict[str, Any]] = None, ) -> Optional[Dict[str, Any]]: """ - Convert a single browser-use action to a semantic workflow step. + Convert a single browser-use action to a semantic workflow step with context. - Mapping: - - navigate/go_to_url → navigation step + Args: + action_type: The type of action (e.g., 'click', 'navigate') + action_dict: The action parameters + element_data: Element data extracted from the DOM + agent_context: Semantic context from the agent's reasoning + + Mapping (browser-use action names): + - navigate → navigation step - input_text → input step with target_text - - click_element → click step with target_text + - click → click step with target_text - send_keys → keypress step - - extract_page_content → extract_page_content step + - extract_content → extract_page_content step - scroll → scroll step """ + agent_context = agent_context or {} # Navigation actions if action_type in ['navigate', 'go_to_url']: - return { + url = action_dict.get('url', '') + step = { 'type': 'navigation', - 'url': action_dict.get('url', ''), - 'description': f'Navigate to {action_dict.get("url", "")}', + 'url': url, + 'description': f'Navigate to {url}', } + # Add semantic metadata for navigation + if agent_context.get('reasoning'): + step['agent_reasoning'] = agent_context['reasoning'] + + return step + # Input text actions elif action_type == 'input_text': target_text = self._extract_target_text(element_data, action_dict) - return { + # Ensure target_text is never empty + if not target_text: + target_text = 'input field' + + step = { 'type': 'input', 'target_text': target_text, 'value': action_dict.get('text', ''), - 'description': f'Enter text into {target_text or "input field"}', + 'description': f'Enter text into {target_text}', } - # Click actions - elif action_type == 'click_element': + # Add element hash for selector population + if element_data and element_data.get('element_hash'): + step['elementHash'] = element_data['element_hash'] + + # Add semantic metadata + if agent_context.get('reasoning'): + step['agent_reasoning'] = agent_context['reasoning'] + if agent_context.get('page_url'): + step['page_context_url'] = agent_context['page_url'] + + return step + + # Click actions (browser-use uses 'click', not 'click_element') + elif action_type in ['click', 'click_element']: target_text = self._extract_target_text(element_data, action_dict) - return { + # Ensure target_text is never empty + if not target_text: + target_text = 'element' + + # Create semantic description + base_description = f'Click on {target_text}' + description = self._create_semantic_description(action_type, base_description, agent_context, target_text) + + step = { 'type': 'click', 'target_text': target_text, - 'description': f'Click on {target_text or "element"}', + 'description': description, } + # Add element hash for selector population + if element_data and element_data.get('element_hash'): + step['elementHash'] = element_data['element_hash'] + + # Add semantic metadata (optional fields that provide context) + if agent_context.get('reasoning'): + step['agent_reasoning'] = agent_context['reasoning'] + if agent_context.get('page_url'): + step['page_context_url'] = agent_context['page_url'] + if agent_context.get('page_title'): + step['page_context_title'] = agent_context['page_title'] + + return step + # Keyboard actions elif action_type == 'send_keys': # For send_keys, we might not have a specific element @@ -165,33 +477,56 @@ def _convert_action_to_step( # Try to get target from last interacted element if available target_text = self._extract_target_text(element_data, action_dict) + # Ensure target_text is never empty + if not target_text: + target_text = 'page' - return { - 'type': 'keypress', + step = { + 'type': 'key_press', 'key': keys, 'target_text': target_text, 'description': f'Press {keys} key', } + # Add element hash for selector population + if element_data and element_data.get('element_hash'): + step['elementHash'] = element_data['element_hash'] + + return step + # Extract content actions - elif action_type == 'extract_page_content': + elif action_type in ['extract_page_content', 'extract_content']: + # Browser-use may use different field names for extraction goal + goal = action_dict.get('value') or action_dict.get('goal') or action_dict.get('content') or 'page content' return { 'type': 'extract_page_content', - 'goal': action_dict.get('value', ''), - 'description': f'Extract: {action_dict.get("value", "")}', + 'goal': goal, + 'description': f'Extract: {goal}', } # Scroll actions elif action_type == 'scroll': - direction = 'down' if action_dict.get('down', True) else 'up' + # Convert browser-use scroll (down: bool, pages: float) to workflow scroll (scrollX, scrollY: int) + # Estimate 800 pixels per page + down = action_dict.get('down', True) pages = action_dict.get('pages', 1.0) - return { + pixels = int(pages * 800) + + step = { 'type': 'scroll', - 'direction': direction, - 'amount': pages, - 'description': f'Scroll {direction} {pages} pages', + 'scrollX': 0, + 'scrollY': pixels if down else -pixels, + 'description': f'Scroll {"down" if down else "up"} {pages} pages', } + # Add semantic metadata + if agent_context.get('reasoning'): + step['agent_reasoning'] = agent_context['reasoning'] + if agent_context.get('page_url'): + step['page_context_url'] = agent_context['page_url'] + + return step + # Dropdown actions - convert to click for now elif action_type == 'select_dropdown_option': target_text = action_dict.get('text', '') @@ -201,13 +536,43 @@ def _convert_action_to_step( 'description': f'Select dropdown option: {target_text}', } + # Navigation actions + elif action_type == 'go_back': + step = { + 'type': 'go_back', + 'description': 'Navigate back to previous page', + } + + # Add semantic metadata + if agent_context.get('reasoning'): + step['agent_reasoning'] = agent_context['reasoning'] + if agent_context.get('page_url'): + step['page_context_url'] = agent_context['page_url'] + + return step + + elif action_type == 'go_forward': + step = { + 'type': 'go_forward', + 'description': 'Navigate forward to next page', + } + + # Add semantic metadata + if agent_context.get('reasoning'): + step['agent_reasoning'] = agent_context['reasoning'] + if agent_context.get('page_url'): + step['page_context_url'] = agent_context['page_url'] + + return step + # Actions we skip or handle differently - elif action_type in ['done', 'switch_tab', 'close_tab']: + elif action_type in ['done', 'switch_tab', 'close_tab', 'write_file', 'replace_file', 'read_file', 'search_google']: return None # These don't translate to workflow steps else: - # Unknown action type - log a warning - print(f'āš ļø Unknown action type: {action_type} - skipping') + # Unknown action type - log a warning (only if not empty) + if action_type: + print(f'āš ļø Unknown action type: {action_type} - skipping') return None def create_workflow_definition( @@ -216,6 +581,7 @@ def create_workflow_definition( description: str, steps: List[Dict[str, Any]], input_schema: Optional[List[Dict[str, Any]]] = None, + version: str = '1.0.0', ) -> Dict[str, Any]: """ Create a complete workflow definition from converted steps. @@ -225,6 +591,7 @@ def create_workflow_definition( description: Workflow description steps: List of converted step dictionaries input_schema: Optional list of input variable definitions + version: Workflow version (default: '1.0.0') Returns: Complete workflow definition dictionary @@ -232,6 +599,7 @@ def create_workflow_definition( return { 'name': name, 'description': description, + 'version': version, 'input_schema': input_schema or [], 'steps': steps, } diff --git a/workflows/workflow_use/healing/service.py b/workflows/workflow_use/healing/service.py index 8c22c45e..ff98e907 100644 --- a/workflows/workflow_use/healing/service.py +++ b/workflows/workflow_use/healing/service.py @@ -121,15 +121,34 @@ def _validate_workflow_quality(self, workflow_definition: WorkflowDefinitionSche def _populate_selector_fields(self, workflow_definition: WorkflowDefinitionSchema) -> WorkflowDefinitionSchema: """Populate cssSelector, xpath, and elementTag fields from interacted_elements_hash_map""" + print('\nšŸ”§ Populating selector fields for workflow steps...') + print(f' Available element hashes in map: {len(self.interacted_elements_hash_map)}') + # Process each step to add back the selector fields - for step in workflow_definition.steps: + populated_count = 0 + for i, step in enumerate(workflow_definition.steps): if isinstance(step, SelectorWorkflowSteps): - if step.elementHash in self.interacted_elements_hash_map: - dom_element = self.interacted_elements_hash_map[step.elementHash] - # DOMInteractedElement has different attribute names - step.cssSelector = getattr(dom_element, 'css_selector', '') or '' - step.xpath = getattr(dom_element, 'x_path', '') or getattr(dom_element, 'xpath', '') - step.elementTag = dom_element.node_name.lower() if hasattr(dom_element, 'node_name') else '' + print(f'\n Step {i + 1} (type={step.type}):') + if hasattr(step, 'elementHash') and step.elementHash: + print(f' elementHash: {step.elementHash}') + if step.elementHash in self.interacted_elements_hash_map: + dom_element = self.interacted_elements_hash_map[step.elementHash] + # DOMInteractedElement has different attribute names + step.cssSelector = getattr(dom_element, 'css_selector', '') or '' + step.xpath = getattr(dom_element, 'x_path', '') or getattr(dom_element, 'xpath', '') + step.elementTag = dom_element.node_name.lower() if hasattr(dom_element, 'node_name') else '' + + print(' āœ… Populated:') + print(f' cssSelector: {step.cssSelector[:80] if step.cssSelector else "(empty)"}') + print(f' xpath: {step.xpath[:80] if step.xpath else "(empty)"}') + print(f' elementTag: {step.elementTag}') + populated_count += 1 + else: + print(' āš ļø elementHash not found in map!') + else: + print(' (no elementHash)') + + print(f'\n āœ… Populated {populated_count} steps with selector fields') # Create the full WorkflowDefinitionSchema with populated fields return workflow_definition @@ -204,6 +223,50 @@ async def _create_workflow_deterministically( # Convert history to steps deterministically steps = self.deterministic_converter.convert_history_to_steps(history_list) + # Transfer element objects from deterministic converter to healing service's map + # This allows _populate_selector_fields to populate cssSelector + # Use the captured element map from the CapturingController instead of history + captured_map = getattr(self, 'captured_element_text_map', {}) + + for history in history_list.history: + if history.model_output is None: + continue + for action in history.model_output.action: + action_dict = action.model_dump() + # Extract index from browser-use action format + for key, value in action_dict.items(): + if isinstance(value, dict) and 'index' in value: + index = value['index'] + if index in self.deterministic_converter.element_hash_map: + element_hash = self.deterministic_converter.element_hash_map[index] + + # First try: Use captured element data (more reliable) + if index in captured_map: + # Create a mock DOMInteractedElement from captured data + captured_data = captured_map[index] + + # Create a simple object with the needed attributes + class MockElement: + def __init__(self, data): + self.node_name = data.get('tag_name', '').upper() + self.css_selector = data.get('css_selector', '') + self.x_path = data.get('xpath', '') + self.xpath = data.get('xpath', '') # Support both attribute names + + mock_element = MockElement(captured_data) + self.interacted_elements_hash_map[element_hash] = mock_element + print(f' šŸ“ Populated selector for hash {element_hash} from captured data (index {index})') + print(f' CSS: {mock_element.css_selector}') + print(f' XPath: {mock_element.x_path}') + continue + + # Fallback: Use history.state.interacted_element + for element in history.state.interacted_element: + if element and hasattr(element, 'highlight_index') and element.highlight_index == index: + self.interacted_elements_hash_map[element_hash] = element + print(f' šŸ“ Populated selector for hash {element_hash} from history (index {index})') + break + # Create workflow definition dict workflow_dict = self.deterministic_converter.create_workflow_definition( name=task, description=f'Workflow for: {task}', steps=steps, input_schema=[] @@ -243,27 +306,155 @@ async def generate_workflow_from_prompt( browser = Browser(use_cloud=use_cloud) - # Note: HealingController's custom action has compatibility issues with current browser-use version - # Using standard Controller for now + # Create a shared map to capture element text during agent execution + element_text_map = {} # Maps index -> {'text': str, 'tag': str, 'xpath': str, etc.} + + # Create a custom controller that captures element mappings from browser_use import Controller + class CapturingController(Controller): + """Controller that captures element text mapping during execution""" + + async def act(self, action, browser_session, *args, **kwargs): + # Get the selector map before action + try: + selector_map = await browser_session.get_selector_map() + + if selector_map: + print(f'šŸ“‹ Captured {len(selector_map)} elements from selector_map') + # selector_map is a dict: {index: DOMInteractedElement} + # We need to extract text/attributes from each element + debug_count = 0 + for index, dom_element in selector_map.items(): + # DEBUG: Print all fields for first 3 elements to see what's available + if debug_count < 3: + print(f'\nšŸ” DEBUG - Element {index}:') + print(f' Type: {type(dom_element)}') + if isinstance(dom_element, dict): + print(f' Dict keys: {list(dom_element.keys())}') + print(f' Content: {dom_element}') + elif hasattr(dom_element, '__dict__'): + print(f' Available fields: {list(dom_element.__dict__.keys())}') + print(f' Values: {dom_element.__dict__}') + else: + attrs = [attr for attr in dir(dom_element) if not attr.startswith('_')] + print(f' Dir (non-private): {attrs}') + # Print values of key attributes + for attr in ['text', 'inner_text', 'node_value', 'node_name', 'attributes']: + if hasattr(dom_element, attr): + val = getattr(dom_element, attr, None) + print(f' {attr}: {val}') + debug_count += 1 + + # Handle dict format (from selector_map) + if isinstance(dom_element, dict): + text = dom_element.get('text', '') + tag_name = dom_element.get('tag_name', '') + attrs = dom_element.get('attributes', {}) + else: + # Extract text by trying multiple field names + text = '' + for text_field in ['text', 'inner_text', 'node_value', 'textContent', 'innerText']: + if hasattr(dom_element, text_field): + text = getattr(dom_element, text_field, '') + if text and text.strip(): + break + tag_name = ( + getattr(dom_element, 'node_name', '').lower() if hasattr(dom_element, 'node_name') else '' + ) + attrs = getattr(dom_element, 'attributes', {}) + + # If still no text, try getting it from attributes + if not text or not text.strip(): + if isinstance(attrs, dict): + # Try common text attributes first + text = ( + attrs.get('aria-label') + or attrs.get('title') + or attrs.get('alt') + or attrs.get('placeholder') + or attrs.get('value') + or '' + ) + + # For anchor tags, try to extract meaningful text from href + if not text and tag_name == 'a' and 'href' in attrs: + href = attrs['href'] + # Extract the last meaningful part of the URL path + # E.g., "https://newsroom.edison.com/releases" -> "releases" + if isinstance(href, str): + # Remove query params and anchors + href = href.split('?')[0].split('#')[0] + # Get the last path segment + path_parts = href.rstrip('/').split('/') + if path_parts: + last_part = path_parts[-1] + # Convert URL-friendly text to readable text + # E.g., "sec-filings" -> "SEC Filings" + if last_part and last_part not in ['www.edison.com', 'edison.com', 'investors']: + text = last_part.replace('-', ' ').replace('_', ' ').title() + print(f' šŸ“Ž Extracted from href: "{text}"') + + # Create a simplified dict with the data we need + # Handle both dict and object formats + if isinstance(dom_element, dict): + element_data = { + 'index': index, + 'tag_name': tag_name or dom_element.get('tag_name', ''), + 'text': text, + 'xpath': dom_element.get('xpath', '') or dom_element.get('x_path', ''), + 'css_selector': dom_element.get('css_selector', ''), + 'attributes': attrs, + } + else: + element_data = { + 'index': index, + 'tag_name': tag_name, + 'text': text, + 'xpath': getattr(dom_element, 'x_path', '') or getattr(dom_element, 'xpath', ''), + 'css_selector': getattr(dom_element, 'css_selector', ''), + 'attributes': attrs, + } + + # Store in the shared map + element_text_map[index] = element_data + + # Show first few captures for debugging + if len(element_text_map) <= 5: + text_preview = element_data['text'][:50] if element_data['text'] else '(no text)' + print(f' Element {index} ({element_data["tag_name"]}): {text_preview}') + + except Exception as e: + print(f'āš ļø Warning: Failed to capture elements before action: {e}') + + # Execute the actual action + result = await super().act(action, browser_session, *args, **kwargs) + return result + agent = Agent( task=prompt, browser_session=browser, llm=agent_llm, page_extraction_llm=extraction_llm, - controller=Controller(), # Using standard controller instead of HealingController + controller=CapturingController(), # Use our custom controller enable_memory=False, max_failures=10, tool_calling_method='auto', ) + # Store the element map for later use + self.captured_element_text_map = element_text_map + # Run the agent to get history + print('šŸŽ¬ Starting agent with element capture enabled...') history = await agent.run() + print(f'āœ… Agent completed. Captured {len(element_text_map)} element mappings total.') # Create workflow definition from the history # Route to deterministic or LLM-based conversion based on flag if self.use_deterministic_conversion: + # Pass the captured element map to the deterministic converter + self.deterministic_converter.captured_element_text_map = element_text_map workflow_definition = await self._create_workflow_deterministically( prompt, history, extract_variables=self.enable_variable_extraction ) diff --git a/workflows/workflow_use/schema/views.py b/workflows/workflow_use/schema/views.py index a78e8aed..1ab41e96 100644 --- a/workflows/workflow_use/schema/views.py +++ b/workflows/workflow_use/schema/views.py @@ -110,6 +110,18 @@ class ScrollStep(BaseWorkflowStep): scrollY: int = Field(..., description='Vertical scroll pixels.') +class GoBackStep(BaseWorkflowStep): + """Navigates back to the previous page in browser history.""" + + type: Literal['go_back'] + + +class GoForwardStep(BaseWorkflowStep): + """Navigates forward to the next page in browser history.""" + + type: Literal['go_forward'] + + class PageExtractionStep(BaseWorkflowStep): """Extracts text from the page using 'page_extraction' (maps to workflow controller's page_extraction).""" @@ -133,6 +145,8 @@ class ExtractStep(BaseWorkflowStep): SelectChangeStep, KeyPressStep, ScrollStep, + GoBackStep, + GoForwardStep, PageExtractionStep, ExtractStep, ] diff --git a/workflows/workflow_use/workflow/semantic_executor.py b/workflows/workflow_use/workflow/semantic_executor.py index cb38c49b..6cf204db 100644 --- a/workflows/workflow_use/workflow/semantic_executor.py +++ b/workflows/workflow_use/workflow/semantic_executor.py @@ -1,5 +1,7 @@ import asyncio +import json import logging +import traceback from typing import TYPE_CHECKING, Dict, List, Optional, Tuple from browser_use import Browser @@ -573,17 +575,26 @@ async def execute_click_step(self, step: ClickStep) -> ActionResult: # Always try to get element_info from semantic mapping for metadata (element_type, etc.) element_info = self._find_element_by_text(step.target_text) - # Try direct selector first (for ID/name attributes) - selector_to_use = await self._try_direct_selector(step.target_text) - - # If direct selector fails, use semantic mapping selector - if not selector_to_use: - if element_info: - selector_to_use = element_info['selectors'] - logger.info(f"Using semantic mapping: '{target_identifier}' -> {selector_to_use}") - else: - # Direct selector worked, but we still want the element_info for metadata - if element_info: + # SEMANTIC WORKFLOW PRIORITY: + # 1. Try direct selector by ID/name (stable, semantic attributes) + # 2. Use hierarchical selector if available and specific enough + # 3. Fall back to CSS selector (which should now include href for links) + + direct_selector = await self._try_direct_selector(step.target_text) + if direct_selector: + selector_to_use = direct_selector + logger.info(f"Using direct selector: '{target_identifier}' -> {selector_to_use}") + elif element_info: + # Use hierarchical selector if it's more specific than the basic selector + hierarchical = element_info.get('hierarchical_selector', '') + basic = element_info.get('selectors', '') + + # Prefer hierarchical if it has positional info (nth-of-type) or IDs + if hierarchical and ('#' in hierarchical or ':nth-of-type' in hierarchical): + selector_to_use = hierarchical + logger.info(f"Using hierarchical selector: '{target_identifier}' -> {selector_to_use}") + else: + selector_to_use = basic logger.info(f"Using semantic mapping: '{target_identifier}' -> {selector_to_use}") elif step.description: @@ -664,10 +675,124 @@ async def click_verifier(): return await self._execute_with_verification_and_retry(click_executor, step, click_verifier) + async def _click_element_by_text_direct(self, target_text: str, element_tag: str = None) -> bool: + """Click element by finding it directly via JavaScript text matching. + + This bypasses selectors entirely and uses the abstract DOM mapping at click time. + """ + page = await self.browser.get_current_page() + + # Build JavaScript to find and click element by text + tag_constraint = f"el.tagName.toLowerCase() === '{element_tag.lower()}'" if element_tag else 'true' + + js_code = f"""() => {{ + const targetText = {json.dumps(target_text)}; + const allElements = document.querySelectorAll('a, button, input, select, textarea, [role="button"], [role="link"], [onclick]'); + + // Try multiple matching strategies in order of specificity + const matches = []; + + for (const el of allElements) {{ + // Skip hidden elements + const rect = el.getBoundingClientRect(); + if (rect.width === 0 || rect.height === 0 || + getComputedStyle(el).visibility === 'hidden' || + getComputedStyle(el).display === 'none') {{ + continue; + }} + + // Check if tag matches (if specified) + if (!({tag_constraint})) {{ + continue; + }} + + // Get visible text from multiple sources + let text = el.textContent?.trim() || ''; + const ariaLabel = el.getAttribute('aria-label') || ''; + const title = el.getAttribute('title') || ''; + const value = el.getAttribute('value') || ''; + const placeholder = el.getAttribute('placeholder') || ''; + + // Combine all text sources + const allText = [text, ariaLabel, title, value, placeholder] + .filter(t => t) + .join(' ') + .toLowerCase(); + + const targetLower = targetText.toLowerCase(); + + // Strategy 1: Exact match (highest priority) + if (allText === targetLower) {{ + matches.push({{ element: el, text: text || ariaLabel || title, priority: 1 }}); + }} + // Strategy 2: Exact word match (split by spaces/special chars) + else if (allText.split(/[\\s\\W]+/).includes(targetLower)) {{ + matches.push({{ element: el, text: text || ariaLabel || title, priority: 2 }}); + }} + // Strategy 3: Contains match + else if (allText.includes(targetLower)) {{ + matches.push({{ element: el, text: text || ariaLabel || title, priority: 3 }}); + }} + // Strategy 4: Fuzzy match (split target into words and match all) + else {{ + const targetWords = targetLower.split(/[\\s\\W]+/).filter(w => w.length > 2); + if (targetWords.length > 0 && targetWords.every(word => allText.includes(word))) {{ + matches.push({{ element: el, text: text || ariaLabel || title, priority: 4 }}); + }} + }} + }} + + // Sort by priority and click the best match + if (matches.length > 0) {{ + matches.sort((a, b) => a.priority - b.priority); + const best = matches[0]; + best.element.click(); + return {{ success: true, clicked: best.text, tag: best.element.tagName, priority: best.priority }}; + }} + + return {{ success: false, error: 'Element not found by text' }}; + }}""" + + try: + result = await page.evaluate(js_code) + + # Handle case where result might be a string instead of dict + if isinstance(result, str): + try: + result = json.loads(result) + except json.JSONDecodeError: + logger.error(f'Failed to parse result as JSON. Got string: {result}') + return False + + # Log the result type and value for debugging + logger.debug(f'Result type: {type(result)}, value: {result}') + + if result and isinstance(result, dict) and result.get('success'): + priority_desc = {1: 'exact', 2: 'word', 3: 'contains', 4: 'fuzzy'} + match_type = priority_desc.get(result.get('priority', 3), 'unknown') + logger.info( + f"āœ… Clicked element by text ({match_type} match): '{target_text}' -> {result.get('tag')}: '{result.get('clicked')}'" + ) + return True + else: + logger.warning(f"āŒ Could not find element by text: '{target_text}'") + return False + except Exception as e: + logger.error(f'Error clicking element by text: {e}') + logger.error(f'Traceback: {traceback.format_exc()}') + return False + async def _click_element_intelligently(self, selector: str, target_text: str, element_info: Dict | None = None) -> bool: """Click element using the most appropriate strategy based on element type.""" page = await self.browser.get_current_page() + # STRATEGY 0: Try direct text-based clicking first (most semantic) + if target_text and target_text.strip(): + element_tag = element_info.get('tag', '').lower() if element_info else None + if await self._click_element_by_text_direct(target_text, element_tag): + return True + logger.info('Direct text click failed, falling back to selector-based approach') + try: # Strategy -1: Check if element_info indicates this is a radio/checkbox, even if selector doesn't show it if element_info and element_info.get('element_type') in ['radio', 'checkbox']: @@ -1435,6 +1560,24 @@ async def execute_scroll_step(self, step: ScrollStep) -> ActionResult: logger.info(msg) return ActionResult(extracted_content=msg, include_in_memory=True) + async def execute_go_back_step(self, step) -> ActionResult: + """Execute go back navigation step.""" + page = await self.browser.get_current_page() + await page.go_back() + + msg = 'šŸ”™ Navigated back to previous page' + logger.info(msg) + return ActionResult(extracted_content=msg, include_in_memory=True) + + async def execute_go_forward_step(self, step) -> ActionResult: + """Execute go forward navigation step.""" + page = await self.browser.get_current_page() + await page.go_forward() + + msg = 'šŸ”œ Navigated forward to next page' + logger.info(msg) + return ActionResult(extracted_content=msg, include_in_memory=True) + async def execute_button_step(self, step) -> ActionResult: """Execute button click step using semantic mapping.""" # Button steps are essentially click steps but with button-specific metadata @@ -1483,6 +1626,10 @@ async def execute_step(self, step: WorkflowStep) -> ActionResult: return await self.execute_button_step(step) elif isinstance(step, ExtractStep): return await self.execute_extract_step(step) + elif step.type == 'go_back': + return await self.execute_go_back_step(step) + elif step.type == 'go_forward': + return await self.execute_go_forward_step(step) else: raise Exception(f'Unsupported step type: {step.type}') diff --git a/workflows/workflow_use/workflow/semantic_extractor.py b/workflows/workflow_use/workflow/semantic_extractor.py index e7fa2fba..644acba8 100644 --- a/workflows/workflow_use/workflow/semantic_extractor.py +++ b/workflows/workflow_use/workflow/semantic_extractor.py @@ -651,7 +651,7 @@ async def extract_interactive_elements(self, page: 'Page') -> List[Dict]: selector = `#${el.id}`; } else { selector = el.tagName.toLowerCase(); - + // Add specific attributes based on widget type const widgetType = containerContext.widget_type; if (widgetType === 'calendar' && el.getAttribute('data-date')) { @@ -661,7 +661,7 @@ async def extract_interactive_elements(self, page: 'Page') -> List[Dict]: } else if (el.getAttribute('data-testid')) { selector += `[data-testid="${el.getAttribute('data-testid')}"]`; } - + // Add other attributes if (el.name) selector += `[name="${el.name}"]`; if (el.type && el.type !== 'submit' && el.type !== 'button' && el.type !== '') { diff --git a/workflows/workflow_use/workflow/service.py b/workflows/workflow_use/workflow/service.py index c87c8363..e0d65ba6 100644 --- a/workflows/workflow_use/workflow/service.py +++ b/workflows/workflow_use/workflow/service.py @@ -110,7 +110,33 @@ async def _run_deterministic_step(self, step: DeterministicWorkflowStep, step_in """Execute a deterministic (controller) action based on step dictionary.""" # Assumes WorkflowStep for deterministic type has 'action' and 'params' keys action_name: str = step.type # Expect 'action' key for deterministic steps - params: Dict[str, Any] = step.model_dump() # Use params if present + all_params: Dict[str, Any] = step.model_dump(exclude_none=True) # Exclude None values to prevent validation errors + + # Filter out workflow metadata fields that shouldn't be passed to browser-use ActionModel + # Note: 'type' is NOT filtered here because some actions (like navigation) need it in their params + workflow_metadata_fields = { + 'description', + 'output', + 'agent_reasoning', + 'page_context_url', + 'page_context_title', + 'cssSelector', + 'xpath', + 'elementTag', + 'elementHash', # These are workflow-specific selector fields + 'target_text', + 'container_hint', + 'position_hint', + 'interaction_type', # Semantic workflow fields + } + + # Keep only action-specific parameters + params = {k: v for k, v in all_params.items() if k not in workflow_metadata_fields} + + # Special handling for actions that use EmptyActionModel (don't accept any parameters) + empty_actions = {'go_back', 'go_forward'} + if action_name in empty_actions: + params = {} ActionModel = self.controller.registry.create_action_model(include_actions=[action_name]) # Pass the params dictionary directly @@ -432,21 +458,34 @@ async def _execute_step(self, step_index: int, step_resolved: WorkflowStep) -> A if isinstance(step_resolved, DeterministicWorkflowStep): from browser_use.agent.views import ActionResult # Local import ok - try: - # Use action key from step dictionary - action_name = step_resolved.type or '[No action specified]' - logger.info(f'Attempting deterministic action: {action_name}') - result = await self._run_deterministic_step(step_resolved, step_index) - if isinstance(result, ActionResult) and result.error: - logger.warning(f'Deterministic action reported error: {result.error}') - raise ValueError(f'Deterministic action {action_name} failed: {result.error}') - except Exception as e: - action_name = step_resolved.type or '[Unknown Action]' - logger.warning( - f'Deterministic step {step_index + 1} ({action_name}) failed: {e}. Attempting fallback with agent.' - ) + # Check if this is a selector step without cssSelector - use semantic execution + action_name = step_resolved.type or '[No action specified]' + requires_selector = action_name in ['click', 'input', 'key_press', 'select_change'] + has_css_selector = hasattr(step_resolved, 'cssSelector') and step_resolved.cssSelector + has_target_text = hasattr(step_resolved, 'target_text') and step_resolved.target_text - raise ValueError(f'Deterministic step {step_index + 1} ({action_name}) failed: {e}') + if requires_selector and not has_css_selector and has_target_text: + # Use semantic execution as fallback + logger.info(f'Step {step_index + 1}: cssSelector missing but target_text present - using semantic execution') + from workflow_use.workflow.semantic_executor import SemanticWorkflowExecutor + + if not hasattr(self, '_semantic_executor'): + self._semantic_executor = SemanticWorkflowExecutor(self.browser, page_extraction_llm=self.page_extraction_llm) + result = await self._semantic_executor.execute_step(step_resolved) + else: + # Use deterministic controller execution + try: + logger.info(f'Attempting deterministic action: {action_name}') + result = await self._run_deterministic_step(step_resolved, step_index) + if isinstance(result, ActionResult) and result.error: + logger.warning(f'Deterministic action reported error: {result.error}') + raise ValueError(f'Deterministic action {action_name} failed: {result.error}') + except Exception as e: + action_name = step_resolved.type or '[Unknown Action]' + logger.warning( + f'Deterministic step {step_index + 1} ({action_name}) failed: {e}. Attempting fallback with agent.' + ) + raise ValueError(f'Deterministic step {step_index + 1} ({action_name}) failed: {e}') # if self.fallback_to_agent: # result = await self._fallback_to_agent(step_resolved, step_index, e)