Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 20 additions & 13 deletions backend/app/runtime/hardware.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,9 +18,10 @@
from app.runtime.supervisor import get_default_engine_dir
from app.schemas.hardware import GpuInfo, HardwareReadiness, StorageInfo

# Reference verified discrete GPU hardware models (M11)
VERIFIED_AMD_MODELS = ["7900", "7800", "6700", "6800"]
VERIFIED_INTEL_MODELS = ["A770", "A750"]
# Reference candidate discrete GPU hardware models (M11)
# Note: These are candidate architectures; physical verification requires real device execution logs.
REFERENCE_AMD_MODELS = ["7900", "7800", "6700", "6800"]
REFERENCE_INTEL_MODELS = ["A770", "A750"]


def detect_nvidia_gpus() -> List[GpuInfo]:
Expand Down Expand Up @@ -111,7 +112,7 @@ def detect_windows_wmi_gpus() -> List[GpuInfo]:

# AMD Radeon Detection
if "RADEON" in name_upper or "AMD" in name_upper:
is_verified = any(v in name_upper for v in VERIFIED_AMD_MODELS)
is_reference = any(v in name_upper for v in REFERENCE_AMD_MODELS)
gpus.append(
GpuInfo(
index=idx,
Expand All @@ -121,13 +122,13 @@ def detect_windows_wmi_gpus() -> List[GpuInfo]:
driver_version=driver_ver,
vendor="amd",
backend="directml",
status_classification="verified" if is_verified else "experimental",
status_classification="candidate_unverified" if is_reference else "experimental",
)
)

# Intel Arc Detection
elif "INTEL" in name_upper and ("ARC" in name_upper or any(m in name_upper for m in ["A770", "A750", "A580", "A380"])):
is_verified = any(v in name_upper for v in VERIFIED_INTEL_MODELS)
is_reference = any(v in name_upper for v in REFERENCE_INTEL_MODELS)
gpus.append(
GpuInfo(
index=idx,
Expand All @@ -137,7 +138,7 @@ def detect_windows_wmi_gpus() -> List[GpuInfo]:
driver_version=driver_ver,
vendor="intel",
backend="directml",
status_classification="verified" if is_verified else "experimental",
status_classification="candidate_unverified" if is_reference else "experimental",
)
)
except (FileNotFoundError, OSError, subprocess.TimeoutExpired):
Expand Down Expand Up @@ -171,7 +172,7 @@ def detect_linux_non_nvidia_gpus() -> List[GpuInfo]:
vram_free_mb=12288,
vendor="amd",
backend="rocm",
status_classification="verified",
status_classification="candidate_unverified",
)
)
except Exception:
Expand Down Expand Up @@ -209,7 +210,7 @@ def detect_all_gpus() -> List[GpuInfo]:
vram_free_mb=12288,
vendor="apple_silicon",
backend="mps",
status_classification="verified",
status_classification="source_compatible_unverified",
)
]

Expand Down Expand Up @@ -304,24 +305,30 @@ async def check_hardware_readiness(engine_dir: Optional[Path] = None) -> Hardwar
guidance.append("SD 1.5 supported in low-vram mode; cloud API inference recommended for larger models.")

elif vendor == "amd":
if status == "verified":
if status == "candidate_unverified":
summary = f"Candidate Architecture (Unverified on Physical Hardware in Current Session): {primary.name} ({primary.vram_total_mb}MB VRAM) using {backend.upper()}."
guidance.append(f"AMD Radeon accelerated via {backend.upper()} with automatic split-cross-attention. Physical generation unverified on this device.")
elif status == "verified":
summary = f"Ready (AMD Reference Hardware): {primary.name} ({primary.vram_total_mb}MB VRAM) using {backend.upper()}."
guidance.append(f"AMD Radeon accelerated via {backend.upper()} with automatic split-cross-attention.")
else:
summary = f"Experimental (AMD): {primary.name} ({primary.vram_total_mb}MB VRAM) detected."
guidance.append("Experimental DirectML acceleration enabled. If performance is inadequate, use Cloud BYOK.")

elif vendor == "intel":
if status == "verified":
if status == "candidate_unverified":
summary = f"Candidate Architecture (Unverified on Physical Hardware in Current Session): {primary.name} ({primary.vram_total_mb}MB VRAM) using {backend.upper()}."
guidance.append(f"Intel Arc accelerated via {backend.upper()}. Physical generation unverified on this device.")
elif status == "verified":
summary = f"Ready (Intel Arc Reference Hardware): {primary.name} ({primary.vram_total_mb}MB VRAM) using {backend.upper()}."
guidance.append(f"Intel Arc accelerated via {backend.upper()}.")
else:
summary = f"Experimental (Intel): {primary.name} detected."
guidance.append("Experimental DirectML acceleration enabled. If performance is inadequate, use Cloud BYOK.")

elif vendor == "apple_silicon":
summary = f"Ready (Apple Silicon): Metal Performance Shaders (MPS) active."
guidance.append("ComfyUI will run natively with Apple Silicon unified memory acceleration.")
summary = "Source-Compatible Architecture (Unverified on Physical Hardware in Current Session): Apple Silicon (MPS)."
guidance.append("ComfyUI and Python service are source-compatible with Apple Silicon MPS; physical macOS package and inference unverified in this session.")

else:
summary = f"Generic GPU detected: {primary.name}."
Expand Down
43 changes: 36 additions & 7 deletions backend/tests/test_m11_hardware.py
Original file line number Diff line number Diff line change
Expand Up @@ -92,7 +92,7 @@ def test_hardware_launch_flags_matrix():

@pytest.mark.asyncio
async def test_readiness_with_amd_reference_hardware(monkeypatch):
"""Readiness assessment correctly classifies verified AMD Radeon reference hardware."""
"""Readiness assessment classifies AMD Radeon reference hardware as candidate_unverified."""
mock_amd = [
GpuInfo(
index=0,
Expand All @@ -101,7 +101,7 @@ async def test_readiness_with_amd_reference_hardware(monkeypatch):
vram_free_mb=14000,
vendor="amd",
backend="directml",
status_classification="verified",
status_classification="candidate_unverified",
)
]
monkeypatch.setattr("app.runtime.hardware.detect_all_gpus", lambda: mock_amd)
Expand All @@ -112,14 +112,14 @@ async def test_readiness_with_amd_reference_hardware(monkeypatch):
assert readiness.has_nvidia_gpu is False
assert readiness.gpu_vendor == "amd"
assert readiness.acceleration_backend == "directml"
assert readiness.status_classification == "verified"
assert readiness.status_classification == "candidate_unverified"
assert readiness.ready_for_local_inference is True
assert "AMD Reference Hardware" in readiness.summary_message
assert "Candidate Architecture" in readiness.summary_message


@pytest.mark.asyncio
async def test_readiness_with_intel_arc_hardware(monkeypatch):
"""Readiness assessment correctly classifies verified Intel Arc reference hardware."""
"""Readiness assessment classifies Intel Arc reference hardware as candidate_unverified."""
mock_intel = [
GpuInfo(
index=0,
Expand All @@ -128,7 +128,7 @@ async def test_readiness_with_intel_arc_hardware(monkeypatch):
vram_free_mb=15000,
vendor="intel",
backend="directml",
status_classification="verified",
status_classification="candidate_unverified",
)
]
monkeypatch.setattr("app.runtime.hardware.detect_all_gpus", lambda: mock_intel)
Expand All @@ -138,9 +138,38 @@ async def test_readiness_with_intel_arc_hardware(monkeypatch):
assert readiness.has_discrete_gpu is True
assert readiness.gpu_vendor == "intel"
assert readiness.acceleration_backend == "directml"
assert readiness.status_classification == "candidate_unverified"
assert readiness.ready_for_local_inference is True
assert "Candidate Architecture" in readiness.summary_message


@pytest.mark.asyncio
async def test_readiness_with_verified_nvidia_hardware(monkeypatch):
"""Readiness assessment classifies NVIDIA CUDA with physical evidence as verified."""
mock_nvidia = [
GpuInfo(
index=0,
name="NVIDIA GeForce RTX 3060",
vram_total_mb=12288,
vram_free_mb=10000,
driver_version="572.70",
vendor="nvidia",
backend="cuda",
status_classification="verified",
)
]
monkeypatch.setattr("app.runtime.hardware.detect_all_gpus", lambda: mock_nvidia)

readiness: HardwareReadiness = await check_hardware_readiness()

assert readiness.has_discrete_gpu is True
assert readiness.has_nvidia_gpu is True
assert readiness.gpu_vendor == "nvidia"
assert readiness.acceleration_backend == "cuda"
assert readiness.status_classification == "verified"
assert readiness.ready_for_local_inference is True
assert "Intel Arc Reference Hardware" in readiness.summary_message
assert "Ready: NVIDIA GeForce RTX 3060" in readiness.summary_message



@pytest.mark.asyncio
Expand Down
2 changes: 1 addition & 1 deletion backend/tests/test_m12_cross_platform.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,7 +36,7 @@ def test_macos_metal_detection(monkeypatch):
assert len(gpus) == 1
assert gpus[0].vendor == "apple_silicon"
assert gpus[0].backend == "mps"
assert gpus[0].status_classification == "verified"
assert gpus[0].status_classification == "source_compatible_unverified"

flags = get_hardware_launch_flags(gpus)
assert "--force-fp16" in flags
Expand Down
9 changes: 9 additions & 0 deletions docs/PLATFORM_PORTABILITY.md
Original file line number Diff line number Diff line change
Expand Up @@ -73,3 +73,12 @@ chmod +x ./berry
# Headless generation
./berry run txt2img --prompt "Cinematic desert landscape" --output ./desert.png
```

---

## 5. Verification Boundaries & Real Evidence Status

- **Windows 11 (x86_64, NVIDIA CUDA)**: **Verified**. Physical runs conducted for Rust launcher build, single-instance mutex, hermetic engine supervision, browser workspace launch, and SD 1.5 local generation via ComfyUI with NVIDIA RTX 3060 12GB.
- **macOS (Apple Silicon, Metal MPS)**: **Source Compatible**. Directory paths, browser launcher commands, Unix domain socket single-instance checks, and Metal fp16 flags implemented and unit-tested; native macOS binary bundling and physical Apple Silicon execution remain unverified in this session.
- **Linux (x86_64, ROCm / CUDA / OneAPI)**: **Source Compatible**. Supervisor process handling, Unix domain socket locking, and headless CLI generation implemented and unit-tested; native Linux binary packaging and physical AMD/Intel hardware execution remain unverified in this session.

29 changes: 15 additions & 14 deletions docs/SUPPORT_MATRIX.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,11 +6,12 @@ This document defines the supported discrete GPU hardware, acceleration backends

## 1. Discrete GPU Tier Classification

| Tier | Definition | Expected Performance | Support Policy |
| Tier | Definition | Expected Performance | Verification Status |
| :--- | :--- | :--- | :--- |
| **Tier 1: Verified Reference Hardware** | Tested and validated directly on physical hardware rigs. Native acceleration with FP16 tensor core optimization. | High (SDXL < 5s, Flux < 15s) | First-class automated testing and bug fixes. |
| **Tier 2: Experimental Hardware** | DirectML / emulated compute support for non-reference discrete GPUs with >= 6GB VRAM. | Moderate (SD1.5 < 4s, SDXL < 15s) | Best-effort community support. Automated fallback flags injected. |
| **Tier 3: Cloud Recommended** | Systems with < 6GB VRAM, legacy architectures (pre-DirectX 12.1), integrated GPUs, or CPU-only setups. | Insufficient for local SDXL/Flux | GUI prompts user to configure Cloud BYOK (OpenAI, Fal.ai, SiliconFlow). |
| **Tier 1: Verified Reference Hardware (Physical Evidence)** | Tested and validated directly on physical hardware rigs with documented generation runs. Native acceleration with FP16 tensor core optimization. | High (SDXL < 5s, Flux < 15s) | **Physically Verified**: NVIDIA GeForce RTX (Ampere/Ada on Windows 11 with CUDA 12.x; verified on RTX 3060 12GB). |
| **Tier 2: Candidate Reference Architecture (Unverified on Physical Hardware)** | Structural detection, launch flags, and DirectML / ROCm / IPEX adapters implemented and tested via mock fixtures. | Moderate (SD1.5 < 4s, SDXL < 15s) | **Candidate Architecture**: AMD Radeon (DirectML/ROCm), Intel Arc (DirectML/OneAPI IPEX). Physical generation unverified in this session. |
| **Tier 3: Source-Compatible Platform (Unverified Native Package)** | Cross-platform code paths (Metal MPS, POSIX sockets, platform launchers) implemented and unit-tested. | Device dependent | **Source Compatible Only**: macOS Apple Silicon, Linux (Ubuntu). Native binary execution and packaging unverified in this session. |
| **Tier 4: Cloud Recommended** | Systems with < 6GB VRAM, legacy architectures (pre-DirectX 12.1), integrated GPUs, or CPU-only setups. | Insufficient for local SDXL/Flux | GUI prompts user to configure Cloud BYOK (OpenAI, Fal.ai, SiliconFlow). |

---

Expand All @@ -19,33 +20,33 @@ This document defines the supported discrete GPU hardware, acceleration backends
### NVIDIA GeForce & RTX (Native CUDA)
| Hardware | Windows 10/11 | Linux (Ubuntu 22.04+) | Backend | Status |
| :--- | :--- | :--- | :--- | :--- |
| RTX 4090 / 4080 / 4070 (Ada) | Native CUDA | Native CUDA | `cuda` | **Verified Reference Hardware** |
| RTX 3090 / 3080 / 3070 / 3060 (Ampere) | Native CUDA | Native CUDA | `cuda` | **Verified Reference Hardware** |
| RTX 2080 / 2070 / 2060 (Turing) | Native CUDA | Native CUDA | `cuda` | **Verified Reference Hardware** |
| RTX 4090 / 4080 / 4070 (Ada) | Native CUDA | Native CUDA | `cuda` | Compatible (Candidate Architecture) |
| RTX 3090 / 3080 / 3070 / 3060 (Ampere) | Native CUDA | Native CUDA | `cuda` | **Physically Verified on Windows Rig** (RTX 3060 12GB, driver 572.70) |
| RTX 2080 / 2070 / 2060 (Turing) | Native CUDA | Native CUDA | `cuda` | Compatible (Candidate Architecture) |
| GTX 1660 / 1650 (6GB) | Native CUDA | Native CUDA | `cuda` (lowvram) | Supported |
| GTX 1080 / 1070 (Pascal) | Native CUDA | Native CUDA | `cuda` (fp32 fallback) | Supported |

### AMD Radeon (DirectML & ROCm)
| Hardware | Windows 10/11 | Linux (Ubuntu 22.04+) | Backend | Status |
| :--- | :--- | :--- | :--- | :--- |
| Radeon RX 7900 XTX / XT (RDNA3) | DirectML (`--directml`) | ROCm 6.1+ | `directml` / `rocm` | **Verified Reference Hardware** |
| Radeon RX 7800 XT / 7700 XT (RDNA3) | DirectML (`--directml`) | ROCm 6.1+ | `directml` / `rocm` | **Verified Reference Hardware** |
| Radeon RX 6800 XT / 6700 XT (RDNA2) | DirectML (`--directml`) | ROCm 5.7+ | `directml` / `rocm` | **Verified Reference Hardware** |
| Radeon RX 7900 XTX / XT (RDNA3) | DirectML (`--directml`) | ROCm 6.1+ | `directml` / `rocm` | Candidate Reference Architecture (Physical execution unverified) |
| Radeon RX 7800 XT / 7700 XT (RDNA3) | DirectML (`--directml`) | ROCm 6.1+ | `directml` / `rocm` | Candidate Reference Architecture (Physical execution unverified) |
| Radeon RX 6800 XT / 6700 XT (RDNA2) | DirectML (`--directml`) | ROCm 5.7+ | `directml` / `rocm` | Candidate Reference Architecture (Physical execution unverified) |
| Radeon RX 6600 / 6500 XT | DirectML (`--lowvram`) | Community ROCm | `directml` | Experimental |
| Older Radeon RX 5000 / Vega | DirectML | Not recommended | `directml` | Experimental / Cloud Recommended |

### Intel Arc Discrete GPUs
| Hardware | Windows 10/11 | Linux (Ubuntu 22.04+) | Backend | Status |
| :--- | :--- | :--- | :--- | :--- |
| Intel Arc A770 (16GB) | DirectML (`--directml`) | OneAPI IPEX | `directml` / `ipex` | **Verified Reference Hardware** |
| Intel Arc A750 (8GB) | DirectML (`--directml`) | OneAPI IPEX | `directml` / `ipex` | **Verified Reference Hardware** |
| Intel Arc A770 (16GB) | DirectML (`--directml`) | OneAPI IPEX | `directml` / `ipex` | Candidate Reference Architecture (Physical execution unverified) |
| Intel Arc A750 (8GB) | DirectML (`--directml`) | OneAPI IPEX | `directml` / `ipex` | Candidate Reference Architecture (Physical execution unverified) |
| Intel Arc A580 / A380 | DirectML (`--lowvram`) | OneAPI IPEX | `directml` | Experimental |

### Apple Silicon (macOS Metal)
| Hardware | macOS 14+ (Sonoma) | macOS 13 (Ventura) | Backend | Status |
| :--- | :--- | :--- | :--- | :--- |
| Apple M1 / M2 / M3 / M4 (Max / Pro) | Native Metal (MPS) | Native Metal (MPS) | `mps` | **Verified Reference Hardware** |
| Apple M1 / M2 / M3 (Base 8GB/16GB) | Native Metal (`--lowvram`) | Native Metal (`--lowvram`) | `mps` | Supported |
| Apple M1 / M2 / M3 / M4 (Max / Pro) | Native Metal (MPS) | Native Metal (MPS) | `mps` | Source Compatible (Native build/run unverified) |
| Apple M1 / M2 / M3 (Base 8GB/16GB) | Native Metal (`--lowvram`) | Native Metal (`--lowvram`) | `mps` | Source Compatible (Native build/run unverified) |

---

Expand Down
Loading