forked from jwvanderstam/LocalChat
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathconfig.py
More file actions
720 lines (596 loc) · 31.9 KB
/
Copy pathconfig.py
File metadata and controls
720 lines (596 loc) · 31.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
"""
Configuration Module
===================
Manages configuration settings and application state for the LocalChat RAG application.
Handles environment variables, database settings, RAG parameters, Ollama GPU settings,
observability options, and persistent state.
Classes:
AppState: Manages persistent application state (active model, document count)
Constants:
Database configuration (PG_HOST, PG_PORT, DB_POOL_MIN_CONN, DB_POOL_MAX_CONN)
Ollama configuration (OLLAMA_BASE_URL, OLLAMA_NUM_GPU)
RAG configuration (CHUNK_SIZE, TOP_K_RESULTS, HYBRID_SEARCH_ENABLED, etc.)
Application settings (SECRET_KEY, UPLOAD_FOLDER, APP_VERSION, etc.)
Observability (METRICS_TOKEN, ENABLE_PERF_METRICS, SLOW_QUERY_THRESHOLD)
Demo mode (DEMO_MODE — disables JWT authentication; never use in production)
Example:
>>> from config import app_state, CHUNK_SIZE, OLLAMA_NUM_GPU
>>> app_state.set_active_model("llama3.2")
>>> print(f"Chunk size: {CHUNK_SIZE}, GPU layers: {OLLAMA_NUM_GPU}")
"""
import json
import os
import secrets
from datetime import datetime
from typing import Any
from dotenv import load_dotenv
from .utils.logging_config import get_logger
# Load environment variables from .env file
load_dotenv()
# Setup logger
logger = get_logger(__name__)
# ============================================================================
# SECURITY CONFIGURATION
# ============================================================================
# Deployment environment — "production", "development", or "" for local
APP_ENV: str = os.environ.get('APP_ENV', '')
# Secret keys - MUST be set in production!
_SECRET_KEY_RAW: str | None = os.environ.get('SECRET_KEY')
if not _SECRET_KEY_RAW or _SECRET_KEY_RAW == 'change-this-to-a-random-secret-key-in-production':
if APP_ENV == 'production':
raise ValueError("SECRET_KEY must be set in production!")
SECRET_KEY: str = secrets.token_hex(32)
logger.warning("Using generated SECRET_KEY - set SECRET_KEY in .env for production")
else:
SECRET_KEY: str = _SECRET_KEY_RAW
_JWT_SECRET_KEY_RAW: str | None = os.environ.get('JWT_SECRET_KEY')
if not _JWT_SECRET_KEY_RAW or _JWT_SECRET_KEY_RAW == 'change-this-to-a-random-jwt-secret-in-production':
if APP_ENV == 'production':
raise ValueError("JWT_SECRET_KEY must be set in production!")
JWT_SECRET_KEY: str = secrets.token_hex(32)
logger.warning("Using generated JWT_SECRET_KEY - set JWT_SECRET_KEY in .env for production")
else:
JWT_SECRET_KEY: str = _JWT_SECRET_KEY_RAW
JWT_ACCESS_TOKEN_EXPIRES: int = int(os.environ.get('JWT_ACCESS_TOKEN_EXPIRES', '3600'))
# Rate limiting settings
RATELIMIT_ENABLED: bool = os.environ.get('RATELIMIT_ENABLED', 'True').lower() == 'true'
RATELIMIT_CHAT: str = str(os.environ.get('RATELIMIT_CHAT', '10 per minute'))
RATELIMIT_UPLOAD: str = str(os.environ.get('RATELIMIT_UPLOAD', '5 per hour'))
RATELIMIT_MODELS: str = str(os.environ.get('RATELIMIT_MODELS', '20 per minute'))
RATELIMIT_GENERAL: str = str(os.environ.get('RATELIMIT_GENERAL', '60 per minute'))
# Redis settings (shared by cache layer and rate limiting storage)
REDIS_ENABLED: bool = os.environ.get('REDIS_ENABLED', 'False').lower() == 'true'
REDIS_HOST: str = str(os.environ.get('REDIS_HOST', 'localhost'))
REDIS_PORT: int = int(os.environ.get('REDIS_PORT', '6379'))
REDIS_PASSWORD: str | None = os.environ.get('REDIS_PASSWORD') or None
# When REDIS_ENABLED=true, REDIS_STRICT=true (default) aborts startup if Redis is
# unreachable rather than silently falling back to in-memory. Set REDIS_STRICT=false
# to restore soft-fallback behaviour (useful for dev boxes that run without Redis).
REDIS_STRICT: bool = os.environ.get('REDIS_STRICT', 'true').lower() == 'true'
# Rate limiting storage URI.
# Uses Redis DB 1 (DB 0 is reserved for application caches) when Redis is
# enabled. Falls back to in-process memory when Redis is not configured.
if REDIS_ENABLED:
_redis_auth = f":{REDIS_PASSWORD}@" if REDIS_PASSWORD else ""
RATELIMIT_STORAGE_URI: str = f"redis://{_redis_auth}{REDIS_HOST}:{REDIS_PORT}/1"
else:
RATELIMIT_STORAGE_URI: str = "memory://"
# CORS settings
CORS_ENABLED: bool = os.environ.get('CORS_ENABLED', 'False').lower() == 'true'
CORS_ORIGINS: list[str] = [o.strip() for o in os.environ.get('CORS_ORIGINS', 'localhost,127.0.0.1').split(',')]
# Admin credentials — legacy env-var admin account + first-start seeding
ADMIN_USERNAME: str = os.environ.get('ADMIN_USERNAME', 'admin')
ADMIN_PASSWORD: str = os.environ.get('ADMIN_PASSWORD', '')
_WEAK_PLACEHOLDERS: frozenset[str] = frozenset({
'secret', 'changeme', 'change-me', 'dev', 'test', 'password', 'admin',
'change-this-to-a-random-secret-key-in-production',
'change-this-to-a-random-jwt-secret-in-production',
})
def validate_secrets() -> None:
"""Abort startup if production secrets are weak or missing.
Enforces: minimum 32-char length for SECRET_KEY / JWT_SECRET_KEY,
rejection of known placeholder values, and a non-empty ADMIN_PASSWORD.
Raises SystemExit(1) so uvicorn startup aborts cleanly.
"""
if APP_ENV != 'production':
return
errors: list[str] = []
if SECRET_KEY.lower() in _WEAK_PLACEHOLDERS:
errors.append("SECRET_KEY is a known placeholder — set a random value")
elif len(SECRET_KEY) < 32:
errors.append("SECRET_KEY is too short (minimum 32 characters)")
if JWT_SECRET_KEY.lower() in _WEAK_PLACEHOLDERS:
errors.append("JWT_SECRET_KEY is a known placeholder — set a random value")
elif len(JWT_SECRET_KEY) < 32:
errors.append("JWT_SECRET_KEY is too short (minimum 32 characters)")
if not ADMIN_PASSWORD:
errors.append("ADMIN_PASSWORD must be set in production")
if errors:
for msg in errors:
logger.critical("[Security] %s", msg)
raise SystemExit(1)
# ============================================================================
# DATABASE CONFIGURATION
# ============================================================================
PG_HOST: str = str(os.environ.get('PG_HOST', 'localhost'))
try:
PG_PORT: int = int(os.environ.get('PG_PORT', '5432'))
except ValueError:
logger.warning("Invalid PG_PORT value, defaulting to 5432")
PG_PORT: int = 5432
PG_USER: str = str(os.environ.get('PG_USER', 'postgres'))
_PG_PASSWORD_RAW: str | None = os.environ.get('PG_PASSWORD')
if not _PG_PASSWORD_RAW:
raise ValueError("PG_PASSWORD must be set in .env file!")
PG_PASSWORD: str = _PG_PASSWORD_RAW
PG_DB: str = str(os.environ.get('PG_DB', 'rag_db'))
# Connection Pool Settings
DB_POOL_MIN_CONN: int = int(os.environ.get('DB_POOL_MIN_CONN', '2'))
DB_POOL_MAX_CONN: int = int(os.environ.get('DB_POOL_MAX_CONN', '10'))
# ============================================================================
# OLLAMA CONFIGURATION
# ============================================================================
OLLAMA_BASE_URL: str = str(os.environ.get('OLLAMA_BASE_URL', 'http://localhost:11434'))
# Number of model layers to offload to GPU(s).
# -1 = all layers on GPU (recommended when GPU VRAM is sufficient).
# 0 = CPU only. Set via OLLAMA_NUM_GPU env var.
OLLAMA_NUM_GPU: int = int(os.environ.get('OLLAMA_NUM_GPU', '-1'))
# Context window size (tokens) sent to Ollama as num_ctx.
# KV-cache VRAM ≈ num_ctx × 0.125 MB for a 7B model.
# 8192 → ~0.5 GB (safe for any 12 GB GPU + any quantization)
# 32768 → ~4.0 GB (requires Q4 model ≤5 GB to avoid CPU offload on 12 GB GPU)
# Override via OLLAMA_NUM_CTX env var to match your GPU + model combination.
OLLAMA_NUM_CTX: int = int(os.environ.get('OLLAMA_NUM_CTX', '8192'))
# HTTP timeout for embedding requests (seconds). A 15MB plain-text file
# produces ~14K chunks; at 512/batch that is ~28 embedding calls. 600 s
# gives the CPU embedder safe headroom for the worst-case file size.
OLLAMA_EMBED_TIMEOUT: int = int(os.environ.get('OLLAMA_EMBED_TIMEOUT', '600'))
# Gunicorn worker timeout (seconds). Must be >= OLLAMA_EMBED_TIMEOUT so a
# worker is never killed mid-embed. 600 s supports up to ~15MB TXT uploads.
GUNICORN_TIMEOUT: int = int(os.environ.get('GUNICORN_TIMEOUT', '600'))
# Preferred chat model selected at startup when no model is already active.
# Falls back to the first available model if this name is not installed.
DEFAULT_MODEL: str = os.environ.get('DEFAULT_MODEL', 'llama3.1')
OLLAMA_EMBEDDING_MODEL: str = os.environ.get('OLLAMA_EMBEDDING_MODEL', '')
# ============================================================================
# RAG CONFIGURATION - OPTIMIZED FOR HIGH QUALITY RESPONSES
# ============================================================================
# Chunking - OPTIMIZED (prevents repetition)
CHUNK_SIZE: int = int(os.environ.get("CHUNK_SIZE", "1200")) # Large chunks for context
CHUNK_OVERLAP: int = int(os.environ.get("CHUNK_OVERLAP", "150")) # 12.5% overlap - industry standard (was 300/25%)
CHUNK_SEPARATORS: list[str] = [
'\n\n\n', # Major section breaks
'\n\n', # Paragraph breaks (primary)
'\n', # Line breaks
'. ', # Sentences
'! ', # Sentences
'? ', # Sentences
'; ', # Clauses
': ', # Lists/definitions
', ', # Phrases
' ', # Words
'' # Character-level (last resort)
]
# Table-specific settings - Keep tables intact
TABLE_CHUNK_SIZE: int = 3000 # Even larger to keep more tables intact
KEEP_TABLES_INTACT: bool = True # Always try to keep tables together
MIN_TABLE_ROWS: int = 3 # Min rows to consider as table
# Retrieval Configuration - OPTIMIZED FOR SYNTHESIS
TOP_K_RESULTS: int = int(os.environ.get("TOP_K_RESULTS", "30")) # Wider candidate pool for reranker
MIN_SIMILARITY_THRESHOLD: float = 0.30 # Slightly higher for quality (was 0.25)
RERANK_RESULTS: bool = True # Always re-rank for precision
RERANK_TOP_K: int = int(os.environ.get("RERANK_TOP_K", "12")) # 12 chunks cover multi-chapter docs (was 4)
# Hybrid Search Configuration
HYBRID_SEARCH_ENABLED: bool = True # Enable semantic + BM25 hybrid search
SEMANTIC_WEIGHT: float = float(os.environ.get("SEMANTIC_WEIGHT", "0.70")) # Semantic similarity weight in hybrid
BM25_ENABLED: bool = True # Enable BM25 keyword matching
# Query Enhancement
QUERY_EXPANSION_ENABLED: bool = False # Expand queries with synonyms (opt-in; English business-domain only)
MAX_QUERY_EXPANSIONS: int = 2 # Add up to 2 related terms
QUERY_MIN_LENGTH: int = 10 # Minimum chars for meaningful query
# Advanced RAG features - MAXIMUM QUALITY
USE_CONTEXTUAL_CHUNKS: bool = True # Include adjacent chunks
CONTEXT_WINDOW_SIZE: int = 2 # 2 chunks before/after (increased from 1)
USE_RECIPROCAL_RANK_FUSION: bool = True # Combine multiple ranking signals
# Re-ranking weights - BALANCED FOR QUALITY
SIMILARITY_WEIGHT: float = 0.50 # Semantic similarity
KEYWORD_WEIGHT: float = 0.20 # Exact term matches (increased)
BM25_WEIGHT: float = 0.20 # BM25 score
POSITION_WEIGHT: float = 0.05 # Early chunks bonus
LENGTH_WEIGHT: float = 0.05 # Chunk length preference
# Diversity filtering
ENABLE_DIVERSITY_FILTER: bool = True # Remove near-duplicate chunks
DIVERSITY_THRESHOLD: float = 0.70 # Jaccard similarity threshold — 0.70 avoids pruning
# domain docs that share terminology across chapters (was 0.50 — too aggressive)
# Context quality enhancement
EMPHASIZE_HIGH_SIMILARITY: bool = True # Mark highest similarity chunks
INCLUDE_CONFIDENCE_SCORES: bool = True # Show confidence in context
# Embedding Cache Configuration
EMBEDDING_CACHE_SIZE: int = 500 # Max cached embeddings
EMBEDDING_CACHE_ENABLED: bool = True # Enable query embedding caching
# Processing Configuration
MAX_WORKERS: int = 8 # Parallel processing threads
BATCH_SIZE: int = 512 # Embeddings batch size
BATCH_MAX_WORKERS: int = 8 # Batch processor workers
# Concurrent sub-batches sent to Ollama during embedding. Ollama serialises
# GPU work internally, so >2 gives diminishing returns and risks starving
# concurrent chat requests on memory-constrained machines.
EMBEDDING_CONCURRENT_BATCHES: int = int(os.environ.get('EMBEDDING_CONCURRENT_BATCHES', '2'))
# Database Performance
# ef_search is computed dynamically in documents.py as max(top_k * 2, 40)
# to balance recall and query latency based on the configured TOP_K_RESULTS.
DB_INDEX_TYPE: str = 'hnsw' # Use HNSW index
# L3 Database Cache Configuration
L3_CACHE_ENABLED: bool = True # Enable DB cache
L3_CACHE_TTL: int = 86400 # 24 hours default
L3_CACHE_TABLE: str = "query_cache" # Cache table name
# Performance Monitoring
ENABLE_PERF_METRICS: bool = True # Enable metrics collection
SLOW_QUERY_THRESHOLD: float = 1.0 # Log queries > 1s
# ============================================================================
# WEB SEARCH CONFIGURATION (RAG Enhanced Mode)
# ============================================================================
WEB_SEARCH_ENABLED: bool = os.environ.get('WEB_SEARCH_ENABLED', 'True').lower() == 'true'
WEB_SEARCH_MAX_RESULTS: int = int(os.environ.get('WEB_SEARCH_MAX_RESULTS', '5'))
WEB_SEARCH_TIMEOUT: int = int(os.environ.get('WEB_SEARCH_TIMEOUT', '10'))
WEB_SEARCH_FETCH_PAGES: bool = os.environ.get('WEB_SEARCH_FETCH_PAGES', 'False').lower() == 'true'
WEB_SEARCH_MAX_PAGE_CHARS: int = int(os.environ.get('WEB_SEARCH_MAX_PAGE_CHARS', '2000'))
# ============================================================================
# TOOL CALLING CONFIGURATION
# ============================================================================
TOOL_CALLING_ENABLED: bool = os.environ.get('TOOL_CALLING_ENABLED', 'True').lower() == 'true'
TOOL_MAX_ROUNDS: int = int(os.environ.get('TOOL_MAX_ROUNDS', '5'))
# ============================================================================
# QUERY PLANNER CONFIGURATION (Feature 2.1)
# ============================================================================
# Set QUERY_PLANNER_ENABLED=false to skip the planning step entirely.
QUERY_PLANNER_ENABLED: bool = os.environ.get('QUERY_PLANNER_ENABLED', 'True').lower() == 'true'
# Set GRAPH_RAG_ENABLED=true to extract named entities on ingest and use graph
# expansion to broaden BM25 query terms. Requires: pip install 'spacy>=3.7.0'
# and: python -m spacy download en_core_web_sm
GRAPH_RAG_ENABLED: bool = os.environ.get('GRAPH_RAG_ENABLED', 'False').lower() == 'true'
# Set LONG_TERM_MEMORY_ENABLED=true to retrieve past memories at query time
# and inject them into the system prompt. Extraction is triggered manually via
# POST /api/memory/extract (or a cron job pointing at that endpoint).
LONG_TERM_MEMORY_ENABLED: bool = os.environ.get('LONG_TERM_MEMORY_ENABLED', 'False').lower() == 'true'
# ============================================================================
# CLOUD MODEL FALLBACK CONFIGURATION (Feature 1.3)
# ============================================================================
# Set CLOUD_FALLBACK_ENABLED=true to enable cloud fallback when the local
# model refuses to answer (e.g. "I don't know" responses).
# Requires litellm: pip install 'litellm>=1.67.0'
CLOUD_FALLBACK_ENABLED: bool = os.environ.get('CLOUD_FALLBACK_ENABLED', 'false').lower() == 'true'
CLOUD_PROVIDER: str = os.environ.get('CLOUD_PROVIDER', '') # e.g. "openai", "anthropic"
CLOUD_API_KEY: str | None = os.environ.get('CLOUD_API_KEY') or None
CLOUD_MODEL: str = os.environ.get('CLOUD_MODEL', '') # e.g. "gpt-4o", "claude-3-5-haiku"
# Phrases that indicate a local refusal (case-insensitive regex alternation).
# Only checked on responses shorter than 500 chars to avoid false positives.
CLOUD_REFUSAL_PATTERNS: list[str] = [
r"I don't know",
r"I cannot",
r"I'm not sure",
r"I don't have information",
r"I don't have access",
r"I'm unable to",
r"no information",
r"not mentioned in",
r"not provided in",
]
# ============================================================================
# MULTI-MODEL ROUTER CONFIGURATION
# ============================================================================
# Set MODEL_ROUTER_ENABLED=true to activate rule-based model selection.
# When a model class is not configured (empty env var) the router falls
# back to the currently active model — zero-risk opt-in.
MODEL_ROUTER_ENABLED: bool = os.environ.get('MODEL_ROUTER_ENABLED', 'false').lower() == 'true'
# Model IDs per class — leave empty to use the active model for that class.
MODEL_FAST: str = os.environ.get('MODEL_FAST', '')
MODEL_BASE: str = os.environ.get('MODEL_BASE', '')
MODEL_LARGE: str = os.environ.get('MODEL_LARGE', '')
MODEL_CODE: str = os.environ.get('MODEL_CODE', '')
MODEL_VISION: str = os.environ.get('MODEL_VISION', '')
# ============================================================================
# AGGREGATOR AGENT CONFIGURATION
# ============================================================================
# Set AGGREGATOR_AGENT_ENABLED=true to route all retrieval through the
# AggregatorAgent instead of the direct pipeline. The agent dispatches
# tools in parallel, retries failures, and deduplicates results.
# Safe to leave false during transition; falls back to the direct path on
# any agent-level error.
AGGREGATOR_AGENT_ENABLED: bool = os.environ.get('AGGREGATOR_AGENT_ENABLED', 'false').lower() == 'true'
# Max retry attempts per tool call (1 = 2 total tries).
AGENT_MAX_RETRIES: int = int(os.environ.get('AGENT_MAX_RETRIES', '1'))
# ============================================================================
# MCP SERVER CONFIGURATION
# ============================================================================
# Set MCP_ENABLED=true to route retrieval and web search through separate
# MCP servers instead of direct in-process imports. Each server runs as its
# own process/container (see docker-compose.yml and mcp_servers/).
# When an MCP server is unreachable its circuit breaker opens and the core
# app falls back to direct imports automatically — zero data loss.
MCP_ENABLED: bool = os.environ.get('MCP_ENABLED', 'false').lower() == 'true'
MCP_LOCAL_DOCS_URL: str = os.environ.get('MCP_LOCAL_DOCS_URL', 'http://localhost:5001')
MCP_WEB_SEARCH_URL: str = os.environ.get('MCP_WEB_SEARCH_URL', 'http://localhost:5002')
MCP_CLOUD_CONNECTORS_URL: str = os.environ.get('MCP_CLOUD_CONNECTORS_URL', 'http://localhost:5003')
MCP_TIMEOUT: int = int(os.environ.get('MCP_TIMEOUT', '30'))
# Circuit breaker: open after N consecutive failures, attempt recovery after M seconds
MCP_CIRCUIT_FAILURE_THRESHOLD: int = int(os.environ.get('MCP_CIRCUIT_FAILURE_THRESHOLD', '5'))
MCP_CIRCUIT_RECOVERY_TIMEOUT: int = int(os.environ.get('MCP_CIRCUIT_RECOVERY_TIMEOUT', '60'))
# ============================================================================
# PLUGIN SYSTEM CONFIGURATION
# ============================================================================
# Set PLUGINS_ENABLED=false to skip plugin loading entirely at startup.
PLUGINS_ENABLED: bool = os.environ.get('PLUGINS_ENABLED', 'True').lower() == 'true'
# Directory (relative to repo root) that is scanned for .py plugin files.
PLUGINS_DIR: str = os.environ.get('PLUGINS_DIR', 'plugins')
# ============================================================================
# LLM CONFIGURATION - MAXIMUM QUALITY
# ============================================================================
DEFAULT_TEMPERATURE: float = 0.0 # ZERO temperature for maximum factuality
MAX_CONTEXT_LENGTH: int = int(os.environ.get("MAX_CONTEXT_LENGTH", str(OLLAMA_NUM_CTX * 3)))
# OLLAMA_NUM_CTX is tokens; multiply by 3 to approximate characters for format_context_for_llm.
# Default leaves ~1/3 of the window for system prompt + history + response.
# ============================================================================
# APPLICATION SETTINGS
# ============================================================================
# Application version (override with APP_VERSION env var for CI-stamped builds)
APP_VERSION: str = os.environ.get('APP_VERSION', '1.0.0')
# Supported file types
SUPPORTED_IMAGE_EXTENSIONS: list[str] = ['.png', '.jpg', '.jpeg', '.gif', '.webp']
SUPPORTED_EXTENSIONS: list[str] = (
['.pdf', '.txt', '.docx', '.md', '.pptx', '.xlsx', '.py', '.js', '.ts', '.eml']
+ SUPPORTED_IMAGE_EXTENSIONS
)
# PDF extraction backend. "auto" tries pymupdf4llm → pdfplumber → pypdf in order;
# set explicitly to "pymupdf4llm", "pdfplumber", or "pypdf" to force one extractor.
PDF_LOADER: str = os.environ.get('PDF_LOADER', 'auto')
# Scheduled re-ingestion — re-process documents older than REINGEST_MAX_AGE_HOURS.
# Disabled by default; set REINGEST_ENABLED=true to activate.
REINGEST_ENABLED: bool = os.environ.get('REINGEST_ENABLED', 'false').lower() == 'true'
REINGEST_MAX_AGE_HOURS: int = int(os.environ.get('REINGEST_MAX_AGE_HOURS', '168')) # 7 days
# Workspace presence — TTL in seconds for an active presence entry.
PRESENCE_TTL_SECONDS: int = int(os.environ.get('PRESENCE_TTL_SECONDS', '30'))
# Graph store backend. "postgres" uses existing entity_relations tables.
# "kuzu" uses the Kuzu embedded graph DB (requires kuzu>=0.6.0 and KUZU_DB_PATH).
GRAPH_BACKEND: str = os.environ.get('GRAPH_BACKEND', 'postgres')
KUZU_DB_PATH: str = os.environ.get('KUZU_DB_PATH', '')
# Vision / multimodal configuration
VISION_DESCRIBE_PROMPT: str = (
"Describe this image in detail. Include all visible text, charts, tables, diagrams, "
"key objects, and any relevant context. Be comprehensive so the description can be "
"used to answer questions about the image."
)
# Flask settings
UPLOAD_FOLDER: str = 'uploads'
MAX_CONTENT_LENGTH: int = int(os.environ.get('MAX_CONTENT_LENGTH', str(16 * 1024 * 1024))) # Default: 16MB
# Abort startup when DB is unavailable — prevent silent degraded-mode starts in prod
REQUIRE_DATABASE: bool = os.environ.get('REQUIRE_DATABASE', 'false').lower() == 'true'
# Logging
LOG_FILE: str = os.environ.get('LOG_FILE', 'logs/app.log')
# Set LOG_FORMAT=json to emit JSON lines (recommended for production log aggregators)
LOG_FORMAT: str = os.environ.get('LOG_FORMAT', 'text')
# State persistence file
STATE_FILE: str = 'app_state.json'
# ============================================================================
# RERANKER CONFIGURATION (Feature 5.2)
# ============================================================================
# Cross-encoder re-ranking of retrieved chunks using sentence-transformers.
# Enabled by default — significantly improves retrieval precision with minimal
# overhead on modern hardware. Set RERANKER_ENABLED=false to disable if CPU
# latency is a concern (e.g. very slow machines or embedded deployments).
# Falls back to cross-encoder/ms-marco-MiniLM-L-6-v2 when no fine-tuned model
# exists at RERANKER_MODEL_PATH; the base model is downloaded automatically
# (~80 MB, requires internet on first start).
RERANKER_ENABLED: bool = os.environ.get('RERANKER_ENABLED', 'true').lower() == 'true'
RERANKER_MODEL_PATH: str = os.environ.get('RERANKER_MODEL_PATH', './models/reranker/latest')
RERANKER_WEIGHT: float = float(os.environ.get('RERANKER_WEIGHT', '0.3'))
FEEDBACK_FINETUNE_MIN_PAIRS: int = int(os.environ.get('FEEDBACK_FINETUNE_MIN_PAIRS', '50'))
# ============================================================================
# MICROSOFT GRAPH / OAUTH CONFIGURATION (Feature 5.3)
# ============================================================================
MICROSOFT_CLIENT_ID: str = os.environ.get('MICROSOFT_CLIENT_ID', '')
MICROSOFT_CLIENT_SECRET: str = os.environ.get('MICROSOFT_CLIENT_SECRET', '')
MICROSOFT_TENANT_ID: str = os.environ.get('MICROSOFT_TENANT_ID', 'common')
MICROSOFT_REDIRECT_URI: str = os.environ.get(
'MICROSOFT_REDIRECT_URI',
'http://localhost:5000/api/oauth/microsoft/callback',
)
# ============================================================================
# GOOGLE DRIVE / OAUTH CONFIGURATION
# ============================================================================
GOOGLE_CLIENT_ID: str = os.environ.get('GOOGLE_CLIENT_ID', '')
GOOGLE_CLIENT_SECRET: str = os.environ.get('GOOGLE_CLIENT_SECRET', '')
GOOGLE_REDIRECT_URI: str = os.environ.get(
'GOOGLE_REDIRECT_URI',
'http://localhost:5000/api/oauth/google/callback',
)
# ============================================================================
# CONFLUENCE CONFIGURATION
# ============================================================================
CONFLUENCE_URL: str = os.environ.get('CONFLUENCE_URL', '')
CONFLUENCE_EMAIL: str = os.environ.get('CONFLUENCE_EMAIL', '')
CONFLUENCE_API_TOKEN: str = os.environ.get('CONFLUENCE_API_TOKEN', '')
# Fernet key for encrypting sensitive text columns at rest. Generate with:
# python -c "from cryptography.fernet import Fernet; print(Fernet.generate_key().decode())"
# TOKEN_ENCRYPTION_KEY is accepted as a legacy alias for backward compatibility.
ENCRYPTION_KEY: str = (
os.environ.get('ENCRYPTION_KEY', '') or os.environ.get('TOKEN_ENCRYPTION_KEY', '')
)
TOKEN_ENCRYPTION_KEY: str = ENCRYPTION_KEY
# ============================================================================
# DEMO MODE
# ============================================================================
# DEMO_MODE=true is intended ONLY for single-user local evaluation.
# It disables JWT authentication and suppresses web search.
# NEVER enable demo mode in a network-accessible or multi-user deployment.
DEMO_MODE: bool = os.environ.get('DEMO_MODE', 'false').lower() == 'true'
if DEMO_MODE:
logger.warning(
"DEMO_MODE is ON — authentication is disabled. "
"Do not expose this instance to untrusted networks."
)
# ============================================================================
# METRICS / OBSERVABILITY
# ============================================================================
# Optional static bearer token that Prometheus (or an operator) must supply
# when scraping /api/metrics. Leave empty to allow unauthenticated access
# (acceptable when the endpoint is behind a firewall or a private network).
METRICS_TOKEN: str = os.environ.get('METRICS_TOKEN', '')
if not METRICS_TOKEN:
logger.warning(
"METRICS_TOKEN is not set — /api/metrics is accessible without authentication. "
"Set METRICS_TOKEN in .env to restrict access."
)
class AppState:
"""
Manages persistent application state.
Handles runtime configuration such as active model and document count,
persisting state to a JSON file for recovery after restarts.
Attributes:
state_file (str): Path to state persistence file
state (Dict[str, Any]): Current application state
Example:
>>> state = AppState()
>>> state.set_active_model("llama3.2")
>>> model = state.get_active_model()
>>> print(model)
llama3.2
"""
def __init__(self, state_file: str = STATE_FILE) -> None:
"""
Initialize application state manager.
Args:
state_file: Path to state persistence file
"""
self.state_file: str = state_file
self.state: dict[str, Any] = self._load_state()
logger.info("Application state initialized")
def _load_state(self) -> dict[str, Any]:
"""
Load state from JSON file.
Returns:
Dictionary containing application state, or default state if file
doesn't exist or cannot be read.
Note:
If loading fails, returns default state and logs the error.
"""
if os.path.exists(self.state_file):
try:
with open(self.state_file) as f:
state = json.load(f)
logger.debug(f"Loaded state from {self.state_file}")
return state
except Exception:
logger.exception("Error loading state")
# Return default state
default_state = {
'active_model': None,
'document_count': 0,
'last_updated': None
}
logger.debug("Using default application state")
return default_state
def _save_state(self) -> None:
"""
Save state to JSON file using an atomic replace so concurrent
gunicorn workers never read a partially-written file.
"""
try:
self.state['last_updated'] = datetime.now().isoformat()
tmp = self.state_file + '.tmp'
with open(tmp, 'w') as f:
json.dump(self.state, f, indent=2)
os.replace(tmp, self.state_file)
logger.debug(f"Saved state to {self.state_file}")
except Exception:
logger.exception("Error saving state")
def get_active_model(self) -> str | None:
"""
Get the currently active model name.
Returns:
Name of active model, or None if no model is set
Example:
>>> model = app_state.get_active_model()
>>> if model:
... print(f"Using model: {model}")
"""
return self.state.get('active_model')
def set_active_model(self, model_name: str) -> None:
"""
Set the active model name.
Args:
model_name: Name of the model to set as active
Example:
>>> app_state.set_active_model("llama3.2:latest")
"""
self.state['active_model'] = model_name
self._save_state()
logger.info("Active model set to: %s", str(model_name).replace('\r', '').replace('\n', ' '))
def get_document_count(self) -> int:
"""
Get the current document count.
Returns:
Number of documents in the system
"""
return self.state.get('document_count', 0)
def set_document_count(self, count: int) -> None:
"""
Set the document count.
Args:
count: New document count
Raises:
ValueError: If count is negative
Example:
>>> app_state.set_document_count(10)
"""
if count < 0:
logger.error(f"Invalid document count: {count}")
raise ValueError("Document count cannot be negative")
self.state['document_count'] = count
self._save_state()
logger.debug(f"Document count set to: {count}")
def increment_document_count(self, increment: int = 1) -> None:
"""
Increment the document count.
Args:
increment: Amount to increment by (default: 1)
Example:
>>> app_state.increment_document_count(5)
"""
current = self.state.get('document_count', 0)
self.state['document_count'] = current + increment
self._save_state()
logger.debug(f"Document count incremented by {increment} to {self.state['document_count']}")
# ============================================================================
# GLOBAL INSTANCES
# ============================================================================
# Create global app state instance
app_state = AppState()
def validate_config() -> None:
"""
Validate configuration values for logical consistency.
Raises:
ValueError: If any configuration value is invalid.
"""
if CHUNK_OVERLAP >= CHUNK_SIZE:
raise ValueError(
f"CHUNK_OVERLAP ({CHUNK_OVERLAP}) must be less than CHUNK_SIZE ({CHUNK_SIZE})"
)
if RERANK_TOP_K > TOP_K_RESULTS:
raise ValueError(
f"RERANK_TOP_K ({RERANK_TOP_K}) cannot exceed TOP_K_RESULTS ({TOP_K_RESULTS})"
)
if OLLAMA_NUM_GPU < -1:
raise ValueError(
f"OLLAMA_NUM_GPU ({OLLAMA_NUM_GPU}) must be >= -1 (-1 means all layers on GPU)"
)
if DB_POOL_MIN_CONN > DB_POOL_MAX_CONN:
raise ValueError(
f"DB_POOL_MIN_CONN ({DB_POOL_MIN_CONN}) cannot exceed DB_POOL_MAX_CONN ({DB_POOL_MAX_CONN})"
)
logger.debug("Configuration validation passed")
validate_config()
logger.info("Configuration module loaded")
logger.debug(f"Database: {PG_HOST}:{PG_PORT}/{PG_DB}")
logger.debug(f"Chunk size: {CHUNK_SIZE}, Overlap: {CHUNK_OVERLAP}")
logger.debug(f"Top-K results: {TOP_K_RESULTS}, Min similarity: {MIN_SIMILARITY_THRESHOLD}")