Repository navigation
Expand file tree
/
Copy pathconfig.example.yaml
More file actions
192 lines (182 loc) · 9.22 KB
/
Copy pathconfig.example.yaml
File metadata and controls
192 lines (182 loc) · 9.22 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
# =============================================================================
# dfe-transform-vector — example configuration
# =============================================================================
#
# Copy to config.yaml and modify as needed.
#
# Configuration cascade (highest to lowest priority):
# 1. CLI args (--config, --log-level, etc.)
# 2. Environment variables (DFE_TRANSFORM_*)
# 3. .env file in the working directory (never a parent's)
# 4. Config file (this file)
# 5. Hard-coded defaults
# -----------------------------------------------------------------------------
# Pipeline identity
# -----------------------------------------------------------------------------
pipeline:
name: "my-pipeline"
# -----------------------------------------------------------------------------
# DFE source name
# -----------------------------------------------------------------------------
# The one setting a working config cannot go without. It derives the platform
# topic names and the consumer group. With pipeline.name "my-pipeline" above:
# source.topics -> ["my_source_land"]
# sink.topic -> "my_source_load"
# source.group_id -> "dfe-transform-vector-my-pipeline"
# Anything set explicitly below wins over the derived value. Without it,
# sink.topic has no default and startup stops at "sink.topic must not be empty".
dfe_source: "my_source"
# -----------------------------------------------------------------------------
# Source: the bus (Kafka topics) or direct (a scalo Push listener)
# -----------------------------------------------------------------------------
# One deployment runs one transport. On `direct` there is no broker: the
# receiver or the fetcher pushes records straight at `listen`, and the
# brokers/topics/group_id below are unused. See templates/bus.yaml and
# templates/direct.yaml for the Vector pipeline each one produces.
# source:
# transport: "bus" # bus | direct
# listen: "0.0.0.0:6000" # direct only: the Push listener address
# brokers:
# - "localhost:9092"
# topics:
# - "raw_events_land"
# group_id: "dfe-transform-vector-my-pipeline"
# decoding:
# codec: "json" # json, raw_bytes, protobuf
# auto_offset_reset: "largest" # largest, smallest
# session_timeout_ms: 30000 # consumer group session timeout
# commit_interval_ms: 5000 # offset commit interval
# drain_timeout_ms: 15000 # max wait during rebalance (default: half session_timeout)
# topic_lag_metric: true # expose consumer lag metric
# # Commit an offset, or answer a push, only once the record is delivered.
# acknowledgements:
# enabled: true
# # Vector reads the credentials from files through its directory secret
# # backend, never from its own config. Point secret_dir at a directory
# # holding `username` and `password` files (a mounted Kubernetes Secret), or
# # set DFE_TRANSFORM_SOURCE_SASL_USERNAME / _PASSWORD instead. A ${VAR}
# # placeholder is refused: Vector expands none.
# sasl:
# enabled: true
# mechanism: "scram_sha_512" # plain, scram_sha_256, scram_sha_512
# secret_dir: "/var/run/secrets/dfe-kafka"
# tls:
# enabled: true
# # DFE_TRANSFORM_KAFKA_SECURITY_PROTOCOL containing SSL (SASL_SSL, SSL)
# # turns this on for source and sink. No value turns it off.
# # librdkafka overrides — production defaults baked in, override here
# # librdkafka_options:
# # fetch.max.bytes: "10485760"
# # queued.min.messages: "100000"
# -----------------------------------------------------------------------------
# Sink: the bus (a Kafka topic) or direct (the next stage's Push listener)
# -----------------------------------------------------------------------------
# `topic` still applies on direct: it is the routing key the next stage picks
# its table by, so a source keeps its name across the hop.
# sink:
# transport: "bus" # bus | direct
# endpoint: "http://dfe-loader:6000" # direct only: where the records go next
# brokers:
# - "localhost:9092"
# topic: "enriched_events_land"
# key_field: ".org_id" # Event field to use as Kafka partition key
# encoding: "json" # json, raw_bytes
# compression: "zstd" # none, gzip, lz4, snappy, zstd
# message_timeout_ms: 0 # 0: an outage holds records, never rejects them
# socket_timeout_ms: 60000 # broker socket timeout
# batch:
# max_events: 10000 # Vector-level batch size (default 10K)
# # max_bytes: 8388608 # optional byte limit
# timeout_secs: 1 # flush timeout
# sasl:
# enabled: true
# mechanism: "scram_sha_512"
# secret_dir: "/var/run/secrets/dfe-kafka"
# tls:
# enabled: true
# # librdkafka overrides — production defaults baked in, override here
# # librdkafka_options:
# # batch.size: "8388608"
# # linger.ms: "20"
# -----------------------------------------------------------------------------
# The two supervisor-to-Vector legs (direct transport only)
# -----------------------------------------------------------------------------
# Vector does not speak scalo's Push protocol, so on `direct` the supervisor
# translates: it hands records to Vector over `to_vector` and takes them back
# over `from_vector`, then pushes them to sink.endpoint. Both live inside one
# pod, so both stay on loopback. Change them only where something else on the
# pod already holds a port.
# bridge:
# to_vector: "127.0.0.1:6100" # Vector's own `vector` source binds here
# from_vector: "127.0.0.1:6101" # the supervisor's listener binds here
# batch_size: 500 # records moved per hop
# -----------------------------------------------------------------------------
# Transform files (user-supplied Vector transform YAML)
# -----------------------------------------------------------------------------
# transforms:
# dir: "/etc/dfe/transforms/"
# # Or specify individual files (loaded in order):
# # files:
# # - "/etc/dfe/transforms/parse.yaml"
# # - "/etc/dfe/transforms/enrich.yaml"
# # - "/etc/dfe/transforms/filter.yaml"
# -----------------------------------------------------------------------------
# Vector subprocess settings
# -----------------------------------------------------------------------------
# vector:
# binary: "/usr/local/bin/vector"
# data_dir: "/var/lib/vector"
# # Where the assembled Vector config is written. The default is the
# # container path; a bare install has no write access there.
# config_dir: "/var/run/vector/config"
# # Vector's API is unauthenticated and nothing here uses it, so it is off.
# # Turned on, it binds a loopback address only.
# api_enabled: false
# api_address: "127.0.0.1:8686"
# log_level: "info"
# # Version pinning — prevents silent version drift
# version: "0.58.0"
# version_check: "strict" # strict, warn, disabled
# # The supervisor runs `binary` and never fetches one, so `preshipped` is
# # the only value accepted. To run a different Vector, point `binary` at it
# # and set `version` to match.
# version_source: "preshipped"
# -----------------------------------------------------------------------------
# Metrics endpoint
# -----------------------------------------------------------------------------
# Also serves /livez and /readyz -- the whole probe surface is on this port,
# and it carries Vector's own vector_* as well: the wrapper scrapes the
# exporter below and merges the samples into this registry. `--metrics-addr`
# and METRICS_ADDR outrank the address set here.
# metrics:
# address: "0.0.0.0:9090"
# # Where Vector's prometheus_exporter binds. Loopback -- the wrapper is its
# # only reader, so nothing outside the pod needs the port.
# vector_metrics_address: "127.0.0.1:9598"
# # Scrape ticks a merged vector_* gauge may go unseen before it is zeroed.
# vector_metrics_expiry_ticks: 4
# # Seconds the sink may hold records without delivering any before /readyz
# # reports not ready. 0 never does.
# sink_stall_secs: 60
# -----------------------------------------------------------------------------
# Logging
# -----------------------------------------------------------------------------
# --log-level/LOG_LEVEL and --log-format/LOG_FORMAT outrank these;
# --verbose/--quiet outrank everything.
logging:
level: "info" # trace, debug, info, warn, error
format: "auto" # json, text, auto (text on a tty)
# -----------------------------------------------------------------------------
# Scaling pressure
# -----------------------------------------------------------------------------
# NOTHING IN THIS APP READS `pressure_threshold`. It is accepted for parity
# with the `scaling:` block every DFE service takes, and range-checked. scalo's
# runtime reads its own `enabled` and `memory_gate_threshold` keys from this
# block for the pressure gauge the subprocess circuit gates.
#
# What actually scales this app is the chart's KEDA ScaledObject -- CPU, plus
# any trigger a deployment adds -- gated by the Vector subprocess circuit, which
# pins pressure to 0 while the subprocess is down (more pods cannot help a dead
# subprocess).
# scaling:
# pressure_threshold: 0.8