-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy path.env.example
More file actions
227 lines (186 loc) · 6.12 KB
/
Copy path.env.example
File metadata and controls
227 lines (186 loc) · 6.12 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
# ============================================
# AVI Configuration Template
# ============================================
# Copy this file to .env and configure for your environment
# Minimal required: MAIN_LLM_API_KEY, MAIN_LLM_MODEL
# ============================================
# Application Settings
# ============================================
APP_NAME=AVI_PoC
ENVIRONMENT=development
DEBUG=true
# ============================================
# Authentication & Authorization
# ============================================
# Set to true in production to require API keys for all protected endpoints
REQUIRE_API_KEY=false
# API key for internal API client (UI, CLI, scripts)
# Leave empty if REQUIRE_API_KEY is false
AVI_API_KEY=
# Base URL for AVI API
AVI_API_BASE=http://localhost:8000
# ============================================
# Main LLM Configuration (REQUIRED)
# ============================================
# These settings control the primary LLM for generating responses
# You MUST set at least MAIN_LLM_API_KEY and MAIN_LLM_MODEL
MAIN_LLM_API_KEY=sk-or-v1-xxxxx
MAIN_LLM_API_BASE=https://openrouter.ai/api/v1
MAIN_LLM_MODEL=openai/gpt-4o-mini
MAIN_LLM_TEMPERATURE=0.7
MAIN_LLM_MAX_TOKENS=2000
# Legacy aliases (backward compatibility)
# EXTERNAL_LLM_API_KEY=sk-or-v1-xxxxx
# OPENROUTER_API_KEY=sk-or-v1-xxxxx
# ============================================
# Safety Mode Configuration
# ============================================
# Options: disabled, llm, remote, hybrid
# - disabled: No safety filtering (default for dev)
# - llm: Use LLM-based safety checks
# - remote: Use external safety microservice
# - hybrid: Combine vector rules + LLM safety
SAFETY_MODE=disabled
# Safety LLM (if SAFETY_MODE=llm or hybrid)
SAFETY_LLM_API_KEY=
SAFETY_LLM_API_BASE=
SAFETY_LLM_MODEL=
SAFETY_LLM_TEMPERATURE=0.1
SAFETY_LLM_MAX_TOKENS=1000
# Safety Service (if SAFETY_MODE=remote)
SAFETY_SERVICE_URL=
SAFETY_SERVICE_TIMEOUT=5.0
# Local Safety Microservice
SAFETY_LOCAL_API_URL=http://localhost:8001
SAFETY_LOCAL_TIMEOUT=5.0
SAFETY_LOCAL_HEALTHCHECK_URL=
# ============================================
# Streaming Guard Configuration
# ============================================
# How to filter streaming responses
# Options: disabled, vector_only, llm_only, hybrid
STREAM_GUARD_MODE=hybrid
# ============================================
# Scoring LLM (Optional)
# ============================================
# Used for evaluating response quality
SCORING_LLM_API_KEY=
SCORING_LLM_API_BASE=
SCORING_LLM_MODEL=
SCORING_LLM_TEMPERATURE=0.0
SCORING_LLM_MAX_TOKENS=10
# ============================================
# RAG & Vector Database
# ============================================
# Similarity threshold for RAG retrieval
RAG_THRESHOLD=0.75
# Vector database provider: chroma or qdrant
VECTOR_DB_PROVIDER=chroma
# Embedding model configuration
EMBEDDING_MODEL=deepvk/USER-bge-m3
INDEX_DIMENSION=1024
# Device for embedding model: 'cpu' or 'cuda' (default: cpu)
# If not set, falls back to DEVICE env var from Docker
EMBEDDING_DEVICE=cpu
# Chroma (file-based, default for local dev)
VECTOR_DB_PATH=./data/indexes/chroma
# Qdrant (recommended for production)
QDRANT_HOST=
QDRANT_PORT=6333
QDRANT_API_KEY=
QDRANT_PATH=./data/indexes/qdrant
# ============================================
# Reranker Configuration
# ============================================
# Cross-encoder reranking for better retrieval
RERANK_ENABLED=true
RERANK_MODEL_NAME=cross-encoder/ms-marco-MiniLM-L-6-v2
RERANK_CANDIDATE_COUNT=15
RERANK_SCORE_THRESHOLD=0.0
RERANK_MAX_LENGTH=512
# Device for reranker model: 'cpu' or 'cuda' (default: cpu)
# If not set, falls back to DEVICE env var from Docker
RERANK_DEVICE=cpu
# ============================================
# Cache Configuration
# ============================================
# Backend: memory (default) or redis
CACHE_BACKEND=memory
CACHE_TTL=3600
# Redis cache (optional, for distributed deployments)
REDIS_URL=
REDIS_HOST=localhost
REDIS_PORT=6379
REDIS_DB=0
REDIS_USERNAME=
REDIS_PASSWORD=
# ============================================
# Filtering & Detection Thresholds (Production-Tuned)
# ============================================
# Content filter thresholds (lowered for production sensitivity)
# Relevance thresholds for filter rule matching (0.0-1.0)
FILTER_DEFAULT_THRESHOLD=0.60
FILTER_FALLBACK_THRESHOLD=0.50
# Vector search configuration
VECTOR_SEARCH_TOP_K=10
VECTOR_SEARCH_SIMILARITY_MIN=0.3
# RAG document retrieval settings
RAG_CANDIDATE_COUNT=5
RAG_RELEVANCE_THRESHOLD=0.5
# Cache performance
CACHE_MAX_SIZE=10000
# ============================================
# Monitoring & Observability
# ============================================
# Prometheus metrics
PROMETHEUS_ENABLED=true
PROMETHEUS_ROUTE=/metrics
METRICS_NAMESPACE=avi
CORRELATION_ID_HEADER=X-Correlation-ID
# OpenTelemetry / Distributed Tracing (Jaeger)
OTEL_ENABLED=false
OTEL_SERVICE_NAME=avi-api
OTEL_EXPORTER_OTLP_ENDPOINT=http://jaeger:4318/v1/traces
OTEL_EXPORTER_OTLP_INSECURE=false
OTEL_EXPORTER_JAEGER_HOST=jaeger
OTEL_EXPORTER_JAEGER_PORT=6831
# ChromaDB Telemetry (disable to suppress warnings)
ANONYMIZED_TELEMETRY=false
CHROMA_CLIENT_AUTH_PROVIDER=
CHROMA_CLIENT_AUTH_CREDENTIALS=
# MLflow experiment tracking
ENABLE_MLFLOW=false
MLFLOW_TRACKING_URI=
MLFLOW_EXPERIMENT_NAME=content_filter_metrics
MLFLOW_RUN_NAME=
# Weights & Biases
ENABLE_WANDB=false
WANDB_PROJECT=
WANDB_ENTITY=
WANDB_RUN_NAME=
# Benchmark tracking backend
# Options: mlflow (default), wandb, none
# Controls which experiment tracker to use for benchmarks
BENCHMARK_TRACKER=mlflow
# ============================================
# Vault Integration (Production)
# ============================================
# HashiCorp Vault for secret management
VAULT_ENABLED=false
VAULT_ADDR=
VAULT_NAMESPACE=
VAULT_AUTH_METHOD=token
VAULT_TOKEN=
VAULT_ROLE_ID=
VAULT_SECRET_ID=
VAULT_MOUNT_POINT=kv
VAULT_SECRETS_PATH=avi/production
# ============================================
# Data Directories
# ============================================
# These are auto-created, no need to change
# DATA_DIR=./data
# RAW_DATA_DIR=./data/raw
# PROCESSED_DATA_DIR=./data/processed
# INDEXES_DIR=./data/indexes
# FEEDBACK_DIR=./data/feedback