-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathconfig.example.yaml
More file actions
145 lines (133 loc) · 4.13 KB
/
Copy pathconfig.example.yaml
File metadata and controls
145 lines (133 loc) · 4.13 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
logging:
# Enable debug mode without log interceptor
debug_mode: true
# List of loggers to suppress in normal mode
suppress_loggers:
- "torch"
- "torch.dynamo"
- "torch._dynamo"
core:
# Global HTTP timeout for connection pool (seconds)
global_http_timeout: 600.0
# Timeout for health check requests (seconds)
health_check_timeout: 5.0
# Retry Configuration for API calls
max_retries: 3
retry_min_wait: 1.0
retry_max_wait: 10.0
multimodal_pipeline:
# Parsing method
parse_method: auto
# Extract images (VLM)
enable_image_processing: true
# Extract tables (VLM/LLM)
enable_table_processing: true
# Extract equations
enable_equation_processing: true
# Context window around multimedia
context_window: 2
# Context unit (page or chunk)
context_mode: page
# Maximum tokens in context
max_context_tokens: 3000
# Include headers in context
include_headers: true
# Include captions for images/tables
include_captions: true
markdown_pipeline:
# Maximum tokens per chunk for docling semantic chunking
max_tokens: 4000
server:
# Server host address
host: "localhost"
# CORS allowed origins
cors_origins: "*"
# JWT signing algorithm
jwt_algorithm: "HS256"
# JWT token expiration time in hours
token_expire_hours: 24
# Guest token expiration time in hours
guest_token_expire_hours: 1
# Enable automatic token renewal
token_auto_renew: true
# Renew token when remaining time < this ratio of total
token_renew_threshold: 0.3
# Comma-separated paths that don't require authentication
whitelist_paths: "/health,/docs,/redoc,/openapi.json,/,/chat,/knowledge-graph,/api,/assets,/auth-status"
minio:
# Maximum file size in bytes to download for Base64 encoding (default 20MB)
max_file_size_bytes: 20971520
llm:
# LLM endpoint (without /chat/completions)
base_url: https://url/v1
# Text model name
model_name: llm
# Enable thinking mode for Qwen models
enable_qwen_thinking: false
# Timeout for LLM API calls (seconds)
timeout: 360.0
vlm:
# VLM endpoint
base_url: https://url/v1
# Visual model name
model_name: vlm
# Enable thinking mode for Qwen-VL models
enable_qwen_thinking: false
# Timeout for VLM API calls (seconds)
timeout: 360.0
embedding:
# Embedding endpoint
base_url: https://url/v1
# Embedding model name
model_name: emb
# Maximum tokens per embedding request
max_tokens: 8192
# Vector dimension (provider-dependent)
dimension: 2048
# Maximum batch size for embeddings (optional, provider-dependent; omit or set to None for universal support)
batch_num: 10
# Enable asymmetric embeddings (text_type='query' for search, text_type='document' for indexing).
asymmetric: false
# Timeout for Embedding API calls (seconds)
timeout: 60.0
rerank:
# Enable/disable reranking service
enabled: false
# Reranker API base URL
base_url: "https://url/v1/reranks"
# Reranker model name
model_name: "rerank-model"
# Maximum documents per request (API limit)
max_docs_per_request: 500
# Maximum tokens per document (API limit)
max_tokens_per_doc: 4000
# Maximum tokens per request (API limit)
max_tokens_per_request: 120000
# Whether the model supports dynamic instructions (asymmetric reranking)
supports_instruct: true
# Minimum score threshold (0.0 - 1.0) for keeping chunks after reranking
min_score: 0.0
retrieval:
# Maximum number of entities/relations retrieved from vector DB
top_k: 60
# Maximum number of text chunks retrieved from vector search (before reranking)
vector_search_top_k: 150
# Maximum number of text chunks kept after reranking
chunk_top_k: 20
# Minimum cosine similarity threshold for vector search (filters out noise)
cosine_threshold: 0.2
# Maximum token budget for the entire prompt context
max_total_tokens: 30000
# Maximum token budget allocated for entities
max_entity_tokens: 6000
# Maximum token budget allocated for relations
max_relation_tokens: 8000
search_strategies:
# Default QA search strategy
qa:
# Enable/disable QA search strategy
enabled: true
# Similarity search strategy
similarity:
# Enable/disable Similarity search strategy
enabled: false