Repository navigation
Expand file tree
/
Copy pathdocker-compose-library.yaml
More file actions
executable file
·164 lines (158 loc) · 6.17 KB
/
Copy pathdocker-compose-library.yaml
File metadata and controls
executable file
·164 lines (158 loc) · 6.17 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
services:
# Lightspeed Stack with embedded OGX (library mode)
lightspeed-stack:
build:
context: .
dockerfile: deploy/lightspeed-stack/Containerfile
platform: linux/amd64
container_name: lightspeed-stack
ports:
- "8080:8080"
depends_on:
mock-mcp:
condition: service_healthy
mock-otel:
condition: service_healthy
mock-guardian:
condition: service_healthy
networks:
- lightspeednet
volumes:
# Mount both config files - lightspeed-stack.yaml should have library mode enabled
- ./lightspeed-stack.yaml:/app-root/lightspeed-stack.yaml:Z
- ./run.yaml:/app-root/run.yaml:Z
- ${GCP_KEYS_PATH:-./tmp/.gcp-keys-dummy}:/opt/app-root/.gcp-keys:ro
# Read-only e2e FAISS fixtures — never use as the live KV_RAG_PATH
- ./tests/e2e/rag:/opt/app-root/src/.llama/storage/.e2e-rag-seed:ro,Z
- ${HF_CACHE_PATH:-./tmp/.hf-cache}:/opt/app-root/src/.cache/huggingface:z
- ./tests/e2e/skills:/app-root/skills:ro,Z
- ./tests/e2e/secrets/mcp-token:/tmp/mcp-token:ro,z
- ./tests/e2e/secrets/invalid-mcp-token:/tmp/invalid-mcp-token:ro,z
# Host copy so seed-restore entrypoint changes apply without rebuilding
- ./scripts/entrypoint.sh:/app-root/entrypoint.sh:ro,z
environment:
# LLM Provider API Keys
- BRAVE_SEARCH_API_KEY=${BRAVE_SEARCH_API_KEY:-}
- TAVILY_SEARCH_API_KEY=${TAVILY_SEARCH_API_KEY:-}
# OpenAI
- OPENAI_API_KEY=${OPENAI_API_KEY}
- E2E_OPENAI_MODEL=${E2E_OPENAI_MODEL:-gpt-4o-mini}
# Azure Entra ID credentials (AZURE_API_KEY is obtained dynamically in Python)
- TENANT_ID=${TENANT_ID:-}
- CLIENT_ID=${CLIENT_ID:-}
- CLIENT_SECRET=${CLIENT_SECRET:-}
# RHAIIS
- RHAIIS_URL=${RHAIIS_URL:-}
- RHAIIS_PORT=${RHAIIS_PORT:-}
- RHAIIS_API_KEY=${RHAIIS_API_KEY:-}
- RHAIIS_MODEL=${RHAIIS_MODEL:-}
# RHEL AI
- RHEL_AI_URL=${RHEL_AI_URL:-}
- RHEL_AI_PORT=${RHEL_AI_PORT:-}
- RHEL_AI_API_KEY=${RHEL_AI_API_KEY:-}
- RHEL_AI_MODEL=${RHEL_AI_MODEL:-}
# VertexAI
- GOOGLE_APPLICATION_CREDENTIALS=${GOOGLE_APPLICATION_CREDENTIALS:-}
- VERTEX_AI_PROJECT=${VERTEX_AI_PROJECT:-}
- VERTEX_AI_LOCATION=${VERTEX_AI_LOCATION:-}
# WatsonX
- WATSONX_BASE_URL=${WATSONX_BASE_URL:-}
- WATSONX_PROJECT_ID=${WATSONX_PROJECT_ID:-}
- WATSONX_API_KEY=${WATSONX_API_KEY:-}
- LITELLM_DROP_PARAMS=true
# AWS Bedrock
- AWS_BEARER_TOKEN_BEDROCK=${AWS_BEARER_TOKEN_BEDROCK:-}
# Enable debug logging if needed
- OGX_LOGGING=${OGX_LOGGING:-}
# FAISS test and inline RAG config
- FAISS_VECTOR_STORE_ID=${FAISS_VECTOR_STORE_ID:-}
- RAG_SEED_DIR=/opt/app-root/src/.llama/storage/.e2e-rag-seed
- KV_RAG_PATH=/tmp/e2e-rag-work/kv_store.db
- PDF_KV_RAG_PATH=/tmp/e2e-rag-work/pdf_kv_store.db
# Prevent HuggingFace Hub update checks (HTTP 429 rate-limiting in CI from parallel jobs).
- HF_HUB_OFFLINE=1
# OpenTelemetry configuration. Export is enabled by default and points at
# the mock OTLP/HTTP collector so the OpenTelemetry delivery E2E test works
# without per-scenario reconfiguration. Override any OTEL_* var (or set
# OTEL_SDK_DISABLED=true) to change or disable export.
- OTEL_EXPORTER_OTLP_ENDPOINT=${OTEL_EXPORTER_OTLP_ENDPOINT:-http://mock-otel:4318}
- OTEL_EXPORTER_OTLP_PROTOCOL=${OTEL_EXPORTER_OTLP_PROTOCOL:-http/protobuf}
- OTEL_SERVICE_NAME=${OTEL_SERVICE_NAME:-lightspeed-stack-e2e}
- OTEL_ANONYMIZATION_SECRET=${OTEL_ANONYMIZATION_SECRET:-lightspeed-stack-otel-anonymization-dev-default}
- OTEL_SDK_DISABLED=${OTEL_SDK_DISABLED:-false}
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:8080/liveness"]
interval: 10s # how often to run the check
timeout: 5s # how long to wait before considering it failed
retries: 3 # how many times to retry before marking as unhealthy
start_period: 15s # time to wait before starting checks (increased for library initialization)
# Mock JWKS server for RBAC E2E tests
mock-jwks:
build:
context: ./tests/e2e/mock_jwks_server
dockerfile: Dockerfile
container_name: mock-jwks
ports:
- "8000:8000"
networks:
- lightspeednet
healthcheck:
test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:8000/health')"]
interval: 5s
timeout: 3s
retries: 3
start_period: 2s
mock-mcp:
build:
context: ./tests/e2e/mock_mcp_server
dockerfile: Dockerfile
container_name: mock-mcp
ports:
- "3000:3000"
networks:
- lightspeednet
healthcheck:
test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:3000/health')"]
interval: 5s
timeout: 3s
retries: 3
start_period: 2s
# Mock Granite Guardian OpenAI-compatible endpoint for shield e2e tests.
# Port 8001 avoids colliding with mock-jwks on 8000.
mock-guardian:
build:
context: ./tests/e2e/mock_guardian_server
dockerfile: Dockerfile
container_name: mock-guardian
ports:
- "127.0.0.1:8001:8001"
networks:
- lightspeednet
healthcheck:
test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:8001/health')"]
interval: 5s
timeout: 3s
retries: 3
start_period: 2s
# Mock OTLP/HTTP collector for OpenTelemetry E2E tests.
# lightspeed-stack exports to it by default (see OTEL_* above) and waits for it
# to be healthy, so telemetry is delivered from startup. The port is bound to
# loopback only so it is not exposed beyond the host running the tests.
mock-otel:
build:
context: ./tests/e2e/mock_otel_collector
dockerfile: Dockerfile
container_name: mock-otel
ports:
- "127.0.0.1:4318:4318"
networks:
- lightspeednet
healthcheck:
test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:4318/health')"]
interval: 5s
timeout: 3s
retries: 3
start_period: 2s
networks:
lightspeednet:
driver: bridge