Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
57 changes: 57 additions & 0 deletions MASK_NORMALIZATION.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,57 @@
# Mask Normalization Implementation (Issue #106)

## Berry Mask Convention

**Canonical Semantic**: Painted regions (opaque white, alpha=255) indicate areas TO BE EDITED by AI. Unpainted regions (transparent, alpha=0) indicate areas TO BE PROTECTED.

This convention is documented in:
- `backend/app/runners/mask_converter.py` module docstring
- Test suite in `backend/tests/test_mask_converter.py`

## Provider-Specific Conversions

### ComfyUI
- **Contract**: LoadImage MASK output = `1 - alpha`
- **Issue**: Painted regions (alpha=255) become 0 (protect), transparent (alpha=0) becomes 1 (edit)
- **Solution**: Invert alpha channel before upload
- **Implementation**: `normalize_mask_for_comfyui()` creates temporary inverted mask

### WebUI
- **Contract**: Grayscale where white=edit, black=protect
- **Match**: Semantically matches Berry convention
- **Solution**: Convert RGBA alpha channel to grayscale L mode
- **Implementation**: `normalize_mask_for_webui()` extracts alpha to grayscale

### OpenAI
- **Contract**: Alpha channel where transparent=edit, opaque=protect
- **Issue**: Inverted from Berry convention
- **Solution**: Invert alpha channel
- **Implementation**: `normalize_mask_for_openai()` inverts alpha

### Fal.ai
- **Contract**: Alpha channel where opaque=edit, transparent=protect
- **Match**: Exactly matches Berry convention
- **Solution**: Validate format only, no conversion needed
- **Implementation**: `normalize_mask_for_fal_ai()` validates image

## Integration Points

1. **ComfyUI**: `creative_runner._run_comfy()` converts mask before upload, cleans up temporary file
2. **WebUI**: `webui_runner._run_inpaint()` converts mask before base64 encoding
3. **Cloud**: `creative_runner._run_cloud()` converts based on provider_id
4. **Validation**: All paths validate source/mask dimension alignment

## Cache Invalidation

Runner version bumped from `0.1.0` to `0.2.0` in `compute_creative_cache_hash()` to prevent reuse of results generated with incorrect mask semantics.

## Test Coverage

13 mask converter tests verify:
- Berry convention documentation
- Provider-specific conversions (ComfyUI, WebUI, OpenAI, Fal.ai)
- Dimension validation and preservation
- Edge cases: partial alpha, fully painted, fully transparent
- File handle management

All tests passing.
44 changes: 38 additions & 6 deletions backend/app/runners/creative_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,13 @@
build_comfy_txt2video_graph,
build_comfy_upscale_graph,
)
from app.runners.mask_converter import (
normalize_mask_for_comfyui,
normalize_mask_for_openai,
normalize_mask_for_webui,
normalize_mask_for_fal_ai,
validate_mask_dimensions,
)
from app.runners.webui_runner import WebUIRunner
from app.runtime.credentials import credentials_manager
from app.schemas.cloud import CloudProviderId
Expand Down Expand Up @@ -175,7 +182,11 @@ def compute_creative_cache_hash(
mask_hash: str = "",
provider_id: Optional[str] = None,
) -> str:
"""Compute deterministic semantic cache hash for a creative action (Invariant #5)."""
"""Compute deterministic semantic cache hash for a creative action (Invariant #5).

Cache version updated for Issue #106: mask semantics normalization.
Old cache entries with version 0.1.0 will not match new 0.2.0 entries.
"""
effective_provider = provider_id or resolve_effective_provider(req)
canonical_payload = {
"action": req.action.value,
Expand All @@ -184,7 +195,7 @@ def compute_creative_cache_hash(
"model": req.model,
"engine_id": req.engine_id,
"provider_id": effective_provider,
"runner_version": "0.1.0",
"runner_version": "0.2.0", # Bumped from 0.1.0 for Issue #106 mask normalization
"width": req.width,
"height": req.height,
"steps": req.steps,
Expand Down Expand Up @@ -456,6 +467,7 @@ async def _run_comfy(
# Upload source image and mask to ComfyUI's input directory if needed
uploaded_image_name: Optional[str] = None
uploaded_mask_name: Optional[str] = None
converted_mask_path: Optional[Path] = None

if input_file and input_file.is_file():
try:
Expand All @@ -470,13 +482,28 @@ async def _run_comfy(

if mask_file and mask_file.is_file():
try:
mask_result = await comfy_client.upload_mask(str(mask_file), subfolder="berry_assets")
# Validate and convert mask for ComfyUI (Issue #106)
# ComfyUI LoadImage MASK output = 1 - alpha, requiring inversion
if input_file:
from PIL import Image
source_img = Image.open(input_file)
validate_mask_dimensions(mask_file, source_img.width, source_img.height)

converted_mask_path = normalize_mask_for_comfyui(mask_file)
mask_result = await comfy_client.upload_mask(str(converted_mask_path), subfolder="berry_assets")
uploaded_mask_name = mask_result["name"]
if mask_result.get("subfolder"):
uploaded_mask_name = f"{mask_result['subfolder']}/{uploaded_mask_name}"
logger.debug(f"Uploaded mask to ComfyUI: {uploaded_mask_name}")
logger.debug(f"Uploaded converted mask to ComfyUI: {uploaded_mask_name}")
except Exception as upload_err:
raise RuntimeError(f"Failed to transfer mask to ComfyUI: {upload_err}") from upload_err
finally:
# Clean up temporary converted mask
if converted_mask_path and converted_mask_path.exists():
try:
converted_mask_path.unlink()
except Exception:
pass

# Build workflow graphs using uploaded filenames
if req.action == CreativeActionType.TXT2IMG:
Expand Down Expand Up @@ -617,17 +644,22 @@ async def _run_cloud(
key = credentials_manager.get_key(CloudProviderId.OPENAI)
if not key:
raise RuntimeError("OpenAI API key missing. Configure it in Cloud Providers (BYOK).")
remote_url = await _call_openai_inpaint(req.prompt, image_bytes, mask_bytes, key)
# Convert mask for OpenAI (Issue #106): transparent = edit
converted_mask_bytes = normalize_mask_for_openai(mask_bytes)
remote_url = await _call_openai_inpaint(req.prompt, image_bytes, converted_mask_bytes, key)
elif plan.provider_id == "fal_ai":
key = credentials_manager.get_key(CloudProviderId.FAL)
if not key:
raise RuntimeError("Fal.ai API key is required for cloud inpainting. Configure it in Cloud Settings (BYOK).")
# Convert mask for Fal.ai (Issue #106): opaque = edit (matches Berry, validate only)
converted_mask_bytes = normalize_mask_for_fal_ai(mask_bytes)
converted_mask_b64 = base64.b64encode(converted_mask_bytes).decode("utf-8")
remote_url = await _call_fal_ai_action(
action="inpaint",
prompt=req.prompt,
api_key=key,
image_b64=image_b64,
mask_b64=mask_b64,
mask_b64=converted_mask_b64,
)
else:
raise ValueError(f"Provider {plan.provider_id} does not support cloud inpainting.")
Expand Down
210 changes: 210 additions & 0 deletions backend/app/runners/mask_converter.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,210 @@
"""
Mask Conversion and Normalization for Inpainting (Issue #106).

Canonical Berry Mask Convention:
- Painted (opaque white, alpha=255) regions indicate areas TO BE EDITED by AI
- Unpainted (transparent, alpha=0) regions indicate areas TO BE PROTECTED (kept unchanged)

This module converts Berry masks to provider-specific formats:
- ComfyUI: LoadImage MASK output computes 1-alpha, requiring inversion
- WebUI: Expects grayscale where white=edit, black=protect (matches Berry)
- OpenAI: Uses alpha channel where transparent=edit (inverted from Berry)
- Fal.ai: Uses alpha channel where opaque=edit (matches Berry)

All conversions preserve source/mask dimensions and coordinate alignment.
"""

from io import BytesIO
from pathlib import Path
from typing import Optional, Tuple

from PIL import Image


def normalize_mask_for_comfyui(mask_path: Path) -> Path:
"""
Convert Berry mask to ComfyUI-compatible format.

ComfyUI's LoadImage MASK output = 1 - alpha, so painted regions (alpha=255)
become 0 (protect) instead of 1 (edit). We must invert the alpha channel.

Args:
mask_path: Path to Berry mask (white painted = edit, transparent = protect)

Returns:
Path to converted mask suitable for ComfyUI LoadImage MASK channel

Raises:
FileNotFoundError: If mask_path doesn't exist
ValueError: If image cannot be processed
"""
if not mask_path.is_file():
raise FileNotFoundError(f"Mask file not found: {mask_path}")

try:
img = Image.open(mask_path).convert("RGBA")
width, height = img.size
pixels = img.load()

# Invert alpha channel: painted (255) -> 0, transparent (0) -> 255
for y in range(height):
for x in range(width):
r, g, b, a = pixels[x, y]
pixels[x, y] = (255, 255, 255, 255 - a)

# Save as temporary converted mask
converted_path = mask_path.with_name(f"{mask_path.stem}_comfy{mask_path.suffix}")
img.save(converted_path, "PNG")
return converted_path

except Exception as e:
raise ValueError(f"Failed to convert mask for ComfyUI: {e}") from e


def normalize_mask_for_webui(mask_bytes: bytes) -> bytes:
"""
Convert Berry mask to WebUI-compatible grayscale format.

WebUI expects grayscale where:
- White (255) = edit region
- Black (0) = protect region

Berry convention already matches this (painted white = edit), so we just
convert alpha to grayscale: alpha channel -> single grayscale channel.

Args:
mask_bytes: Berry mask PNG bytes

Returns:
Converted grayscale mask PNG bytes

Raises:
ValueError: If image cannot be processed
"""
try:
img = Image.open(BytesIO(mask_bytes)).convert("RGBA")
width, height = img.size

# Create grayscale image where painted regions are white
grayscale = Image.new("L", (width, height), 0)
pixels_src = img.load()
pixels_dst = grayscale.load()

for y in range(height):
for x in range(width):
_, _, _, a = pixels_src[x, y]
# Alpha 255 (painted) -> white 255 (edit)
# Alpha 0 (transparent) -> black 0 (protect)
pixels_dst[x, y] = a

output = BytesIO()
grayscale.save(output, "PNG")
return output.getvalue()

except Exception as e:
raise ValueError(f"Failed to convert mask for WebUI: {e}") from e


def normalize_mask_for_openai(mask_bytes: bytes) -> bytes:
"""
Convert Berry mask to OpenAI-compatible format.

OpenAI expects alpha channel where:
- Transparent (alpha=0) = edit region
- Opaque (alpha=255) = protect region

This is inverted from Berry convention, so we invert the alpha channel.

Args:
mask_bytes: Berry mask PNG bytes

Returns:
Converted mask PNG bytes with inverted alpha

Raises:
ValueError: If image cannot be processed
"""
try:
img = Image.open(BytesIO(mask_bytes)).convert("RGBA")
width, height = img.size
pixels = img.load()

# Invert alpha: painted (255) -> 0 (edit), transparent (0) -> 255 (protect)
for y in range(height):
for x in range(width):
r, g, b, a = pixels[x, y]
pixels[x, y] = (r, g, b, 255 - a)

output = BytesIO()
img.save(output, "PNG")
return output.getvalue()

except Exception as e:
raise ValueError(f"Failed to convert mask for OpenAI: {e}") from e


def normalize_mask_for_fal_ai(mask_bytes: bytes) -> bytes:
"""
Convert Berry mask to Fal.ai-compatible format.

Fal.ai expects alpha channel where:
- Opaque (alpha=255) = edit region
- Transparent (alpha=0) = protect region

This matches Berry convention, so no conversion needed - just validate format.

Args:
mask_bytes: Berry mask PNG bytes

Returns:
Original mask bytes (already in correct format)

Raises:
ValueError: If image cannot be processed
"""
try:
# Validate image can be loaded
img = Image.open(BytesIO(mask_bytes))
img.verify()
return mask_bytes

except Exception as e:
raise ValueError(f"Failed to validate mask for Fal.ai: {e}") from e


def validate_mask_dimensions(mask_path: Path, source_width: int, source_height: int) -> Tuple[int, int]:
"""
Validate mask dimensions match source image dimensions.

Args:
mask_path: Path to mask image
source_width: Expected width from source image
source_height: Expected height from source image

Returns:
Tuple of (mask_width, mask_height)

Raises:
FileNotFoundError: If mask doesn't exist
ValueError: If dimensions don't match
"""
if not mask_path.is_file():
raise FileNotFoundError(f"Mask file not found: {mask_path}")

try:
with Image.open(mask_path) as img:
mask_width, mask_height = img.size

if mask_width != source_width or mask_height != source_height:
raise ValueError(
f"Mask dimensions ({mask_width}x{mask_height}) do not match "
f"source dimensions ({source_width}x{source_height}). "
"Source and mask must have identical dimensions for proper alignment."
)

return mask_width, mask_height

except ValueError:
raise
except Exception as e:
raise ValueError(f"Failed to validate mask dimensions: {e}") from e
17 changes: 15 additions & 2 deletions backend/app/runners/webui_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@

from app.schemas.creative import CreativeActionRequest, CreativeActionType
from app.storage.asset_store import asset_store
from app.runners.mask_converter import normalize_mask_for_webui, validate_mask_dimensions

logger = logging.getLogger(__name__)

Expand Down Expand Up @@ -222,8 +223,20 @@ async def _run_inpaint(self, client: httpx.AsyncClient, req: CreativeActionReque
if not img_rec or not mask_rec:
raise ValueError("Input or mask asset missing on disk")

img_b64 = base64.b64encode(asset_store.get_absolute_path(img_rec).read_bytes()).decode("utf-8")
mask_b64 = base64.b64encode(asset_store.get_absolute_path(mask_rec).read_bytes()).decode("utf-8")
img_path = asset_store.get_absolute_path(img_rec)
mask_path = asset_store.get_absolute_path(mask_rec)

# Validate mask dimensions match source (Issue #106)
from PIL import Image
source_img = Image.open(img_path)
validate_mask_dimensions(mask_path, source_img.width, source_img.height)

# Convert mask for WebUI (Issue #106): grayscale white=edit, black=protect
mask_bytes = mask_path.read_bytes()
converted_mask_bytes = normalize_mask_for_webui(mask_bytes)

img_b64 = base64.b64encode(img_path.read_bytes()).decode("utf-8")
mask_b64 = base64.b64encode(converted_mask_bytes).decode("utf-8")

payload = {
"init_images": [img_b64],
Expand Down
Loading
Loading