-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathsettings.sample.yaml
More file actions
124 lines (115 loc) · 5.93 KB
/
Copy pathsettings.sample.yaml
File metadata and controls
124 lines (115 loc) · 5.93 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
# Copy to settings.yaml and fill in values. settings.yaml is the primary config
# (environment variables can override individual fields).
LOG_LEVEL: info
PORT: 3000
MON_PORT: 8888
GRPC_PORT: 3001
ENABLE_PPROF: false
MAX_REQUEST_DURATION: 30s
# Per-JWT-subject in-flight request cap; 0 disables it (opt-in — see settings.go).
MAX_CONCURRENT_REQUESTS_PER_SUBJECT: 0
# Process-wide in-flight cap (HTTP+gRPC), the pool admission control (H11). 0 disables.
MAX_CONCURRENT_REQUESTS: 0
# DuckLake is the only query backend. CATALOG_DSN + DATA_PATH are REQUIRED
# (boot fails fast if the DSN is empty): a Postgres connection string in prod, or
# a local catalog-file path for single-node dev.
DUCKLAKE_CATALOG_DSN: ""
DUCKLAKE_DATA_PATH: ""
# Read-only reader role: attach the catalog (+ an optional read replica)
# read-only. Leave false + empty on the materializer/writer.
DUCKLAKE_READ_ONLY: false
DUCKLAKE_CATALOG_READ_DSN: ""
# Embedded DuckDB query engine knobs (all optional; empty/0 keeps defaults).
DUCKDB_MEMORY_LIMIT: ""
DUCKDB_THREADS: 0
DUCKDB_MAX_CONNS: 0
DUCKDB_EXTENSION_DIR: ""
DUCKDB_TEMP_DIRECTORY: ""
DUCKDB_S3_ENDPOINT: ""
# Materializer (post-fact decode loop: din raw_events -> decoded lake tables).
# Enable on exactly one single-writer replica.
MATERIALIZER_ENABLED: false
MATERIALIZER_POLL_INTERVAL: 15s
MATERIALIZER_WORKERS: 0
# Rollup flush cadence during a continuous drain; empty = poll interval (B2).
MATERIALIZER_ROLLUP_INTERVAL: ''
# How often lake.raw_types_latest (the availableCloudEventTypes rollup, dq#40)
# is fully rebuilt — one whole-table raw_events pass per rebuild. Empty = 15m.
MATERIALIZER_RAW_TYPES_INTERVAL: ''
# Backfill tuning for a large one-time catch-up (initial load / long downtime):
# skips the cross-batch dedup anti-join and flushes signals_latest once on
# catch-up. Set for the initial backfill, then unset for steady state.
MATERIALIZER_BACKFILL_MODE: false
# Read only to REJECT sharding (> 1 is refused — single writer only).
MATERIALIZER_SHARD_COUNT: 1
# Daily signals_latest refresh (dq#55): off | on, default on (empty means on).
# on: the once-daily watermarked refresh is THE maintainer of
# lake.signals_latest (the per-pass fold was removed in step 5). off leaves
# the rollup unmaintained — tests/one-off ops only, warned at boot. The
# retired shadow mode is invalid; shadow-era configs must move to on (a
# leftover lake.signals_latest_daily is promoted automatically at first
# boot). Pair on with LAKE_ROLLUP_DAILY_SERVING=true on the query fleet.
MATERIALIZER_DAILY_ROLLUP_MODE: 'on'
# How long after UTC midnight the daily refresh waits (cursor settle +
# straggler margin). Empty = 3h30m, the traffic trough.
MATERIALIZER_DAILY_ROLLUP_DELAY: ''
# Extended KV serving for allLatest + availableSignals (dq#55 step 3):
# off | shadow | serve, dark-launched independently of LATEST_KV_READ_MODE
# (which must not be off when this is set). Query fleet only.
LATEST_KV_READ_MODE_EXTENDED: 'off'
# True whenever the materializer runs MATERIALIZER_DAILY_ROLLUP_MODE=on (the
# default): summaries then serve the exact (rollup ∪ tail) union — a plain
# rollup read under-counts the post-watermark tail. Query fleet.
LAKE_ROLLUP_DAILY_SERVING: true
# Decoded-row retention (Go duration, e.g. 8760h); empty disables pruning.
LAKE_DECODED_RETENTION: ""
# Rebuild signals_latest from full base on boot (disaster recovery only; O(history)).
LAKE_REBUILD_ROLLUP_ON_BOOT: false
# signals-latest NATS KV cache (write side; materializer-only). When enabled the
# materializer folds every decoded signal batch into the KV bucket so
# signalsLatest can be served without per-query DuckLake planning. Best-effort:
# a NATS outage means a staler cache, never a blocked decode.
NATS_URL: ""
LATEST_KV_WRITE_ENABLED: false
# Bucket name; empty = "signals-latest". Writer and reader must agree.
LATEST_KV_BUCKET: ""
# Query-fleet read mode for the cache: off | shadow (serve from the rollup but
# compare against the cache and count dq_lake_latest_kv_shadow_total) | serve
# (answer from the cache, rollup fallback on miss/error). Needs NATS_URL.
LATEST_KV_READ_MODE: off
# Re-run the lake.signals_latest -> KV bootstrap even though its completion
# marker is present (repair after a sustained publish outage). One boot, then unset.
LATEST_KV_FORCE_BOOTSTRAP: false
# Materializer: assert that the bucket mirrors lake.signals_latest completely,
# so query pods may answer a cache MISS as "no data" instead of paying a rollup
# read that returns nothing. One small heartbeat write every 30s, plus a full
# rollup->KV reconcile at boot when the previous writer did not exit cleanly.
# Required for LATEST_KV_NEGATIVE to do anything.
LATEST_KV_COVERAGE: false
# Query fleet: what a cache MISS means. off | shadow (still read the rollup, but
# count what answering "no data" would have gotten wrong —
# dq_lake_latest_kv_false_negative_total must be zero before serve) | serve
# (answer the miss directly). Requires LATEST_KV_READ_MODE=serve, and only takes
# effect while the materializer's coverage assertion is live.
LATEST_KV_NEGATIVE: off
# Require a valid DIMO JWT on the fetch gRPC port. False admits a missing token
# (invalid tokens are always rejected); pair with a NetworkPolicy until true.
FETCH_GRPC_REQUIRE_JWT: false
TOKEN_EXCHANGE_JWK_KEY_SET_URL: http://127.0.0.1:3050/keys
TOKEN_EXCHANGE_ISSUER_URL: xpp
# Bucket holding din's externalized (>inline-threshold) cloudevent blob payloads,
# required for externalized-payload reads (the fetch path downloads/presigns blobs
# from here; the parquet lake lives separately at DUCKLAKE_DATA_PATH).
BLOB_BUCKET: ""
S3_AWS_REGION: ""
S3_AWS_ACCESS_KEY_ID: ""
S3_AWS_SECRET_ACCESS_KEY: ""
# Custom S3 endpoint (full URL) for the blob presign/download path; e.g. MinIO.
# Leave empty for AWS. Mirror in DUCKDB_S3_ENDPOINT for the lake path.
S3_ENDPOINT: ""
IDENTITY_API_URL: ""
# DIMO registry chain + NFT contracts for vendor-module DID construction.
DIMO_REGISTRY_CHAIN_ID: 0
VEHICLE_NFT_ADDRESS: ""
AFTERMARKET_NFT_ADDRESS: ""
SYNTHETIC_NFT_ADDRESS: ""