-
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathconfig.toml
More file actions
197 lines (163 loc) · 6.9 KB
/
Copy pathconfig.toml
File metadata and controls
197 lines (163 loc) · 6.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
# MemoryOps Configuration
# Secrets (API keys, passwords) are NEVER stored here.
# Always use environment variables for secrets (see .env.example).
[server]
host = "0.0.0.0"
port = 8080
request_timeout_secs = 30
# Allow tools/skills endpoint URLs to resolve to loopback/private IP addresses.
# Useful for development or private VPC deployments. Off by default.
allow_private_ips = false
[database]
# Connection string read from DATABASE_URL env var.
# These settings control the connection pool.
max_connections = 20
min_connections = 2
connect_timeout_secs = 5
[redis]
# Connection string read from REDIS_URL env var.
pool_size = 10
# ── Embedding Provider ───────────────────────────────────────────────────────
[embedding]
# Options: "fastembed" (local, default) | "openai"
provider = "fastembed"
model = "BAAI/bge-small-en-v1.5"
# Uncomment to use OpenAI embeddings (requires OPENAI_API_KEY env var)
# [embedding.openai]
# model = "text-embedding-3-small"
# ── LLM Provider ─────────────────────────────────────────────────────────────
[llm]
# Options: "ollama" (local, default) | "openai" | "anthropic"
provider = "ollama"
model = "llama3"
# ⚠️ Docker users: "localhost" does NOT reach the host machine from inside a
# container. Use "host.docker.internal" instead (works on Docker Desktop
# for Mac/Windows; on Linux add --add-host=host.docker.internal:host-gateway
# to your docker run / compose service, or use the host network mode).
#
# Local dev (bare metal / WSL): base_url = "http://localhost:11434"
# Docker Desktop (Mac / Windows): base_url = "http://host.docker.internal:11434"
base_url = "http://host.docker.internal:11434"
timeout_secs = 3
# Ollama Cloud / hosted Ollama (e.g. https://api.ollama.ai):
# Set base_url to your cloud endpoint and uncomment the block below.
# api_key_env names the environment variable that holds your Bearer token —
# the secret itself is never stored in this file.
#
# [llm.ollama]
# api_key_env = "OLLAMA_API_KEY"
# Uncomment to use OpenAI (requires OPENAI_API_KEY env var)
# [llm.openai]
# model = "gpt-4o-mini"
# Uncomment to use Anthropic (requires ANTHROPIC_API_KEY env var)
# [llm.anthropic]
# model = "claude-3-haiku-20240307"
# ── OpenAI-Compatible Providers ───────────────────────────────────────────────
# These providers share the /chat/completions request/response format.
# Set provider = "openrouter", "huggingface", or "openai_compatible".
# Omit base_url for openrouter/huggingface to use the defaults injected automatically.
#
# OpenRouter example:
# [llm]
# provider = "openrouter"
# model = "openai/gpt-4o-mini"
# timeout_secs = 30
#
# [llm.openai_compatible]
# api_key_env = "OPENROUTER_API_KEY"
# headers = { "HTTP-Referer" = "https://your-app.example", "X-OpenRouter-Title" = "MemoryOps" }
#
# Hugging Face Inference Router example:
# [llm]
# provider = "huggingface"
# model = "meta-llama/Llama-3.1-8B-Instruct"
# timeout_secs = 30
#
# [llm.openai_compatible]
# api_key_env = "HF_API_KEY"
#
# Generic OpenAI-compatible endpoint (e.g. vLLM, LM Studio, LocalAI):
# [llm]
# provider = "openai_compatible"
# model = "my-model"
# base_url = "http://localhost:8000/v1"
# timeout_secs = 60
#
# [llm.openai_compatible]
# api_key_env = "MY_PROVIDER_API_KEY" # omit section if no auth is needed
# ── Processing Pipeline ──────────────────────────────────────────────────────
[processor]
# Fast path: max concurrency for synchronous processing
fast_path_concurrency = 4
# Slow path: number of Redis consumer workers
slow_path_workers = 2
# Max retry attempts before sending a job to the DLQ
max_retries = 3
# DLQ entry TTL in days
dlq_ttl_days = 7
# Reprocess a stuck 'processing' event if no heartbeat/update occurs for this many seconds
processing_stale_threshold_secs = 600
# UTC hour for daily maintenance pass (decay/pruning/hard-delete)
maintenance_window_hour_utc = 2
# UTC hour for weekly Sunday promotion pass
decay_window_hour_utc = 3
# ── Promotion Pipeline ───────────────────────────────────────────────────────
[promotion]
# How often to run the promotion scheduler
cadence_minutes = 15
# Episodic events older than this are eligible for clustering
clustering_window_hours = 24
# Minimum mean importance score for a cluster to be promoted
promotion_threshold = 0.65
# ── Decay & Pruning ──────────────────────────────────────────────────────────
[decay]
# Pruning scheduler runs daily at this UTC hour
schedule_hour_utc = 2
# Batch size for pruning to avoid table locks
batch_size = 1000
[decay.semantic]
decay_rate = 0.98
prune_threshold = 0.10
[decay.episodic]
decay_rate = 0.95
prune_threshold = 0.10
# ── Rate Limiting ────────────────────────────────────────────────────────────
[rate_limit]
# Default limits per workspace per minute (configurable per workspace via API)
retrieve_rpm = 60
ingest_rpm = 300
api_rpm = 120
# High-frequency dashboard polling routes get their own bucket so they cannot
# exhaust the workspace's general API quota.
dashboard_rpm = 600
# ── Retrieval Engine ─────────────────────────────────────────────────────────
[retrieval]
# Cosine similarity threshold above which a candidate is considered a duplicate
dedup_threshold = 0.92
# Default token budget if not specified in request
default_token_budget = 4096
# Tokenizer: cl100k_base (GPT-4 compatible)
tokenizer = "cl100k_base"
# Default scoring weights — must sum to 1.0
[retrieval.weights]
semantic_similarity = 0.35
importance = 0.25
recency = 0.20
source_authority = 0.10
memory_type = 0.10
# Default source authority scores
[retrieval.source_authority]
github = 0.9
slack = 0.6
jira = 0.7
linear = 0.7
# ── Observability ────────────────────────────────────────────────────────────
[telemetry]
# Log format: "json" (production) | "pretty" (development)
log_format = "pretty"
# OTEL exporter: "none" | "otlp" | "prometheus"
otel_exporter = "none"
# Slow span warning threshold in milliseconds
slow_span_threshold_ms = 500
# Retrieval trace retention in days
trace_retention_days = 30