Coverage for /usr/local/lib/python3.10/site-packages/opal_server-0.0.0-py3.10.egg/opal_server/config.py: 98%
105 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 11:54 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 11:54 +0000
1import os
2import pathlib
3from enum import Enum
5from opal_common.authentication.types import EncryptionKeyFormat
6from opal_common.confi import Confi
7from opal_common.schemas.data import DEFAULT_DATA_TOPIC, ServerDataSourceConfig
8from opal_common.schemas.webhook import GitWebhookRequestParams
10confi = Confi(prefix="OPAL_")
13class PolicySourceTypes(str, Enum):
14 Git = "GIT"
15 Api = "API"
18class PolicyBundleServerType(str, Enum):
19 HTTP = "HTTP"
20 AWS_S3 = "AWS-S3"
23class ServerRole(str, Enum):
24 Primary = "primary"
25 Secondary = "secondary"
28class OpalServerConfig(Confi):
29 # ws server
30 OPAL_WS_LOCAL_URL = confi.str(
31 "WS_LOCAL_URL",
32 "ws://localhost:7002/ws",
33 description="The local WebSocket URL for OPAL",
34 )
35 OPAL_WS_TOKEN = confi.str(
36 "WS_TOKEN", "THIS_IS_A_DEV_SECRET", description="The WebSocket token for OPAL"
37 )
38 CLIENT_LOAD_LIMIT_NOTATION = confi.str(
39 "CLIENT_LOAD_LIMIT_NOTATION",
40 None,
41 description="If supplied, rate limit would be enforced on server's websocket endpoint. "
42 + "Format is `limits`-style notation (e.g '10 per second'), "
43 + "see link: https://limits.readthedocs.io/en/stable/quickstart.html#rate-limit-string-notation",
44 )
45 # The URL for the backbone pub/sub server (e.g. Postgres, Kfaka, Redis) @see
46 BROADCAST_URI = confi.str(
47 "BROADCAST_URI", None, description="The URL for the backbone pub/sub server"
48 )
49 # The name to be used for segmentation in the backbone pub/sub (e.g. the Kafka topic)
50 BROADCAST_CHANNEL_NAME = confi.str(
51 "BROADCAST_CHANNEL_NAME",
52 "EventNotifier",
53 description="The name to be used for segmentation in the backbone pub/sub",
54 )
55 BROADCAST_RECONNECT_ENABLED = confi.bool(
56 "BROADCAST_RECONNECT_ENABLED",
57 True,
58 description="Reconnect the broadcaster reader on a backbone disconnect instead "
59 "of dropping all client connections. Set to False to revert to the legacy "
60 "(non-reconnecting) broadcaster.",
61 )
62 BROADCAST_RECONNECT_MAX_RETRIES = confi.int(
63 "BROADCAST_RECONNECT_MAX_RETRIES",
64 0,
65 description="Maximum consecutive broadcaster reconnect attempts before giving "
66 "up and letting the worker restart (0 = retry forever).",
67 )
68 BROADCAST_RECONNECT_BACKOFF_MIN_SECONDS = confi.float(
69 "BROADCAST_RECONNECT_BACKOFF_MIN_SECONDS",
70 0.5,
71 description="Minimum backoff in seconds between broadcaster reconnect attempts.",
72 )
73 BROADCAST_RECONNECT_BACKOFF_MAX_SECONDS = confi.float(
74 "BROADCAST_RECONNECT_BACKOFF_MAX_SECONDS",
75 30.0,
76 description="Maximum backoff in seconds between broadcaster reconnect attempts.",
77 )
78 BROADCAST_REPLAY_BUFFER_SIZE = confi.int(
79 "BROADCAST_REPLAY_BUFFER_SIZE",
80 10000,
81 description="Max number of outbound broadcasts buffered while the backbone is "
82 "down and replayed on reconnect (0 disables buffering). On overflow the oldest "
83 "buffered broadcasts are dropped; the resync on reconnect still reconciles clients.",
84 )
85 BROADCAST_RESYNC_ON_RECONNECT = confi.bool(
86 "BROADCAST_RESYNC_ON_RECONNECT",
87 True,
88 description="After a backbone gap that may have lost updates, force this "
89 "worker's connected clients to reconnect so they re-fetch full policy + data "
90 "state (guarantees cross-instance consistency).",
91 )
92 BROADCAST_RESYNC_SETTLE_SECONDS = confi.float(
93 "BROADCAST_RESYNC_SETTLE_SECONDS",
94 2.0,
95 description="Grace period after a broadcaster reconnect before replaying "
96 "buffered broadcasts and resyncing clients, to let peer servers re-subscribe.",
97 )
98 BROADCAST_HEALTHCHECK_ENABLED = confi.bool(
99 "BROADCAST_HEALTHCHECK_ENABLED",
100 True,
101 description="Make /healthcheck reflect the broadcaster reader's health so a "
102 "k8s readiness/liveness probe can route away from or restart a worker whose "
103 "reader is wedged while clients depend on it. Set to False to revert "
104 "/healthcheck to always returning ok.",
105 )
106 BROADCAST_FREEZE_ON_DISCONNECT = confi.bool(
107 "BROADCAST_FREEZE_ON_DISCONNECT",
108 True,
109 description="During a broadcaster backbone gap, freeze client-facing publishes on "
110 "every worker instead of applying them locally — so a write that cannot fan out to "
111 "the whole fleet is never served by just one worker (fleet consistency over "
112 "freshness). Recovery is the reconnect resync: clients re-fetch their configured "
113 "data sources and policy, so updates covered by those are reconciled; one-off "
114 "updates outside them (inline data payloads, ad-hoc fetch URLs) are DROPPED by a "
115 "freeze, not deferred. Internal coordination topics (statistics, keepalive, the git "
116 "webhook trigger) are exempt and keep the deliver-locally + buffer-for-replay "
117 "behavior. Engages only with BROADCAST_RECONNECT_ENABLED (without the reconnecting "
118 "broadcaster there is no gap signal, so the flag is a no-op) and requires "
119 "BROADCAST_RESYNC_ON_RECONNECT (if the resync is disabled, freezing is refused "
120 "with a warning). Set to False for "
121 "the previous behavior, where the receiving worker's own clients update immediately "
122 "and peers only after reconnect.",
123 )
125 # server security
126 AUTH_PRIVATE_KEY_FORMAT = confi.enum(
127 "AUTH_PRIVATE_KEY_FORMAT",
128 EncryptionKeyFormat,
129 EncryptionKeyFormat.pem,
130 description="The format of the private key for authentication",
131 )
132 AUTH_PRIVATE_KEY_PASSPHRASE = confi.str(
133 "AUTH_PRIVATE_KEY_PASSPHRASE",
134 None,
135 description="The passphrase for the private key",
136 )
138 AUTH_PRIVATE_KEY = confi.delay(
139 lambda AUTH_PRIVATE_KEY_FORMAT=None, AUTH_PRIVATE_KEY_PASSPHRASE="": confi.private_key(
140 "AUTH_PRIVATE_KEY",
141 default=None,
142 key_format=AUTH_PRIVATE_KEY_FORMAT,
143 passphrase=AUTH_PRIVATE_KEY_PASSPHRASE,
144 description="The private key for authentication",
145 )
146 )
148 AUTH_JWKS_URL = confi.str(
149 "AUTH_JWKS_URL",
150 "/.well-known/jwks.json",
151 description="The URL for the JSON Web Key Set (JWKS)",
152 )
153 AUTH_JWKS_STATIC_DIR = confi.str(
154 "AUTH_JWKS_STATIC_DIR",
155 os.path.join(os.getcwd(), "jwks_dir"),
156 description="The directory for static JWKS files",
157 )
159 AUTH_MASTER_TOKEN = confi.str(
160 "AUTH_MASTER_TOKEN", None, description="The master token for authentication"
161 )
163 # policy source watcher
164 POLICY_SOURCE_TYPE = confi.enum(
165 "POLICY_SOURCE_TYPE",
166 PolicySourceTypes,
167 PolicySourceTypes.Git,
168 description="Set your policy source can be GIT / API",
169 )
170 POLICY_REPO_URL = confi.str(
171 "POLICY_REPO_URL",
172 None,
173 description="Set your remote repo URL e.g:https://github.com/permitio/opal-example-policy-repo.git\
174 , relevant only on GIT source type",
175 )
176 POLICY_BUNDLE_URL = confi.str(
177 "POLICY_BUNDLE_URL",
178 None,
179 description="Set your API bundle URL, relevant only on API source type",
180 )
181 POLICY_REPO_CLONE_PATH = confi.str(
182 "POLICY_REPO_CLONE_PATH",
183 os.path.join(os.getcwd(), "regoclone"),
184 description="Base path to create local git folder inside it that manage policy change",
185 )
186 POLICY_REPO_CLONE_FOLDER_PREFIX = confi.str(
187 "POLICY_REPO_CLONE_FOLDER_PREFIX",
188 "opal_repo_clone",
189 description="Prefix for the local git folder",
190 )
191 POLICY_REPO_REUSE_CLONE_PATH = confi.bool(
192 "POLICY_REPO_REUSE_CLONE_PATH",
193 False,
194 description="Set if OPAL server should use a fixed clone path (and reuse if it already exists) instead of randomizing its suffix on each run",
195 )
196 POLICY_REPO_MAIN_BRANCH = confi.str(
197 "POLICY_REPO_MAIN_BRANCH",
198 "master",
199 description="The main branch of the policy repository",
200 )
201 POLICY_REPO_SSH_KEY = confi.str(
202 "POLICY_REPO_SSH_KEY", None, description="The SSH key for the policy repository"
203 )
204 POLICY_REPO_MANIFEST_PATH = confi.str(
205 "POLICY_REPO_MANIFEST_PATH",
206 "",
207 description="Path of the directory holding the '.manifest' file (new fashion), or of the manifest file itself (old fashion). Repo's root is used by default",
208 )
209 POLICY_REPO_CLONE_TIMEOUT = confi.int(
210 "POLICY_REPO_CLONE_TIMEOUT",
211 0,
212 description="The timeout for cloning the policy repository (0 means wait forever)",
213 )
214 SCOPES_GIT_FETCH_TIMEOUT = confi.float(
215 "SCOPES_GIT_FETCH_TIMEOUT",
216 120.0,
217 description="Soft timeout in seconds for a single scope git clone/fetch: "
218 "the awaiting operation is abandoned (the event loop and the sync pass "
219 "move on) and the op is logged and skipped, retried next cycle. It is a "
220 "SOFT timeout: the underlying git call keeps running on its own thread, "
221 "and the pinned libgit2 sets no socket/server read timeout, so a "
222 "black-holed remote can keep that thread — and the source's in-flight "
223 "marker — alive for the life of the process (that source is then skipped "
224 "until it recovers or the process restarts); SCOPES_GIT_MAX_ZOMBIES "
225 "bounds how many such threads accumulate. Either way one unreachable repo "
226 "never blocks boot or other scopes (0 = no timeout).",
227 )
228 SCOPES_GIT_MAX_WORKERS = confi.int(
229 "SCOPES_GIT_MAX_WORKERS",
230 # Worst-case OS-thread count during an outage is this limit plus the
231 # number of lingering timed-out ("zombie") ops: a timed-out op releases
232 # its concurrency slot but keeps a daemon thread running on its own.
233 # SCOPES_GIT_MAX_ZOMBIES bounds that tail.
234 10,
235 description="Maximum number of concurrent scope git operations. It bounds "
236 "phase 1 (the network clone/fetch of each distinct repo) AND phase 2 (the "
237 "local change-check of scopes that reuse an already-cloned repo), so it "
238 "sets how many scopes are synced at once in either phase. A timed-out "
239 "operation stops counting against this limit, so one hung remote does not "
240 "hold a concurrency slot — its lingering daemon thread persists on its own "
241 "(a black-holed remote's can persist for the life of the process). That "
242 "tail is bounded by SCOPES_GIT_MAX_ZOMBIES, which is a GLOBAL ceiling: "
243 "read its description, because at that ceiling new git ops are refused "
244 "for every scope, healthy ones included.",
245 )
246 SCOPES_GIT_PRELOAD_DRAIN_TIMEOUT = confi.float(
247 "SCOPES_GIT_PRELOAD_DRAIN_TIMEOUT",
248 10.0,
249 description="Max seconds the pre-fork scope preload waits for in-flight git "
250 "ops to finish before tearing down and forking workers. Ops still lingering "
251 "past this bound are left running on their daemon threads and their cached "
252 "handles are left unfreed by the pre-fork cache reset, avoiding a use-after-free "
253 "(0 = don't wait).",
254 )
255 SCOPES_GIT_MAX_ZOMBIES = confi.int(
256 "SCOPES_GIT_MAX_ZOMBIES",
257 # 4x the default SCOPES_GIT_MAX_WORKERS. Once this many git ops (live +
258 # lingering timed-out) hold a daemon thread, new ops are refused until
259 # threads drain, bounding worst-case thread growth during an outage.
260 40,
261 description="Maximum number of in-flight scope git operations (live plus "
262 "lingering timed-out) allowed to hold a daemon thread at once, counted "
263 "GLOBALLY across all sources. It is a last-resort ceiling on thread "
264 "growth when remotes hang, not a per-source guard — that is handled "
265 "separately, by skipping a source that already has an operation in "
266 "flight. Once the ceiling is reached, new git ops are refused (logged, "
267 "and retried next cycle) for EVERY scope, healthy ones included, until "
268 "enough threads drain; with remotes that never return, that state can "
269 "persist. Set it well above SCOPES_GIT_MAX_WORKERS and alert on the "
270 "refusal log (0 = no cap; a negative value is clamped to 0 and also "
271 "means no cap, at the cost of unbounded thread growth during an "
272 "outage).",
273 )
274 SCOPES_GIT_BACKOFF_BASE_SECONDS = confi.float(
275 "SCOPES_GIT_BACKOFF_BASE_SECONDS",
276 10.0,
277 description="First delay before the periodic sync pass re-attempts a "
278 "SOURCE whose git clone or fetch just failed (unreachable host, revoked "
279 "credentials, deleted repo); every further consecutive failure doubles "
280 "it — 10s, 20s, 40s, ... minutes, hours, days — with no upper bound "
281 "unless SCOPES_GIT_BACKOFF_MAX_SECONDS is set. A delay shorter than the "
282 "gap to the next pass simply does not skip that pass, so the first few "
283 "doublings cost one attempt per pass exactly as before and the schedule "
284 "bites from roughly the fourth consecutive failure. It exists because "
285 "nothing else records a failure: without it every pass re-attempts every "
286 "dead repo, and so does every duplicate scope sharing that repo — a "
287 "handful of dead repositories can account for thousands of clone attempts "
288 "per hour. Only the periodic pass and the boot preload honour it (in "
289 "both phases, and re-checked under the source lock, so duplicates of a "
290 "source that fails in a pass cost one attempt, not one per scope): an "
291 "explicit POST /scopes/:scope_id/refresh, POST /scopes/refresh or "
292 "PUT /scopes attempts the source immediately, so an operator who has "
293 "just repaired credentials recovers at once, and any successful clone or "
294 "fetch clears both the delay and the consecutive-failure count. The state "
295 "is in-memory and per process — it resets on restart. When a periodic "
296 "pass runs (POLICY_REFRESH_INTERVAL > 0) a forked worker inherits "
297 "whatever the pre-fork preload recorded, so the leader does not re-hammer "
298 "repos that already failed at boot; without one, the boot sync is the "
299 "only pass-originated sync, so it drops the inherited entries and "
300 "attempts every source once. Watch "
301 "opal_server.scopes.sources_in_backoff (gauge of sources currently being "
302 "skipped, tagged by pid) and opal_server.scopes.git_op_skipped with "
303 "reason:backoff (counter); the WARNING logged when a source enters "
304 "backoff, and again when its delay first exceeds a day, names the "
305 "repository. 0 or negative disables the backoff entirely — nothing is "
306 "recorded and nothing is skipped; nan and inf are treated as disabled "
307 "too, because they parse cleanly rather than failing the process at "
308 "startup and neither is a duration.",
309 )
310 SCOPES_GIT_BACKOFF_MAX_SECONDS = confi.float(
311 "SCOPES_GIT_BACKOFF_MAX_SECONDS",
312 0.0,
313 description="Optional cap on the per-source doubling backoff of "
314 "SCOPES_GIT_BACKOFF_BASE_SECONDS. 0 (the default), negative, nan or inf "
315 "means NO cap: a repository that has been unreachable for a day is "
316 "checked again in two, then four, and before long only at the next "
317 "restart or explicit refresh — a repository that keeps failing is, in "
318 "all likelihood, dead. Set a positive value to bound instead how stale "
319 "a repository that comes back on its own (without anyone touching its "
320 "scope) can get: the periodic pass then re-attempts it at most that "
321 "long after the previous attempt. A value below the base is floored at "
322 "the base (one pass at a time), never inert. Lowering the cap at "
323 "runtime is not retroactive for delays already armed.",
324 )
325 SCOPES_POLICY_CLONE_WAIT_SECONDS = confi.float(
326 "SCOPES_POLICY_CLONE_WAIT_SECONDS",
327 20.0,
328 description="How long GET /scopes/:scope_id/policy holds a request while "
329 "that scope's clone is still being populated, before falling through to "
330 "the 503 + Retry-After it answers today. The route re-checks once a "
331 "second and returns the bundle the moment the clone is usable. It exists "
332 "because the opal-client PDP ignores Retry-After: it makes five attempts "
333 "with random-exponential backoff capped at 10s (~20-40s of coverage) and "
334 "then stays quiet until the next pub/sub policy message or a reconnect, "
335 "so a clone that outlives those attempts strands that PDP with no policy "
336 "— and the update-all published when a clone completes names only the "
337 "scope that was syncing, so siblings sharing the same clone are not "
338 "woken. Holding the request converts that gap into latency the client "
339 "already tolerates: five client attempts against a 20s hold cover about "
340 "two minutes of clone time, so short and medium re-clones — meaning the "
341 "download phase; the rmtree-and-init window before it answers 503 + "
342 "Retry-After 5 and is not waited on — produce no client-visible gap. "
343 "What this bounds is the WAIT plus at most one more bundle attempt: "
344 "time queued behind other bundle builds on the shared executor is "
345 "outside the deadline, which is what SCOPES_POLICY_CLONE_WAIT_MAX_INFLIGHT "
346 "bounds instead. The budget matters in both directions — 20s is well "
347 "under the 60s ALB idle timeout (a longer hold surfaces as a 504, which "
348 "the client cannot tell apart from a dead server) and far under the "
349 "client's 300s aiohttp total timeout. Readiness is derived from DISK (the "
350 "clone still has no remote-tracking refs), never from an in-process "
351 "marker, so every worker answers alike: the clone runs in the leader "
352 "while this route is served by any worker. The hold is an awaited sleep "
353 "loop, so it occupies no thread between polls and leaves the event loop "
354 "and the gunicorn worker heartbeat unaffected; it is abandoned early if "
355 "the client disconnects. 0 or negative disables the wait (answer 503 "
356 "immediately). nan, inf and -inf are treated as disabled too: unlike a "
357 "non-numeric value, which fails this process at startup when the "
358 "environment is parsed, they parse cleanly — inf would otherwise mean "
359 "the clamped maximum hold on every clone-in-progress request, and nan "
360 "is not a budget at all. Values above 55s are clamped to 55s so the "
361 "hold can never outlive the load balancer's idle timeout.",
362 )
363 SCOPES_POLICY_CLONE_WAIT_MAX_INFLIGHT = confi.int(
364 "SCOPES_POLICY_CLONE_WAIT_MAX_INFLIGHT",
365 64,
366 description="Maximum number of requests one worker process may hold at "
367 "once inside the SCOPES_POLICY_CLONE_WAIT_SECONDS wait. Requests beyond "
368 "the cap get the immediate 503 + Retry-After 30 they would have got "
369 "before the wait existed, so the cap can never make things worse than "
370 "not waiting. It exists because polling is cheap but RELEASING is not: "
371 "when the clone lands, every held request builds a full bundle on the "
372 "loop's DEFAULT executor: about min(32, cpu+4) threads shared by every "
373 "off-loop call this process makes, in front of an unbounded queue. "
374 "Measured throughput there falls from ~52 bundles/s at 32 concurrent "
375 "builds to ~18/s at 1000. Scope git clone/fetch does NOT share that "
376 "pool — each op runs on its own single-use daemon-thread executor, "
377 "bounded by SCOPES_GIT_MAX_WORKERS through a semaphore — so this key is "
378 "the only bound on how many bundle builds can pile up at once. "
379 "Size it so the cap divided by the "
380 "bundles-per-second that pod can really build fits inside the 60s ALB "
381 "idle timeout MINUS the hold: above that, released requests queue past "
382 "the timeout and 504 — the very failure the hold exists to prevent — and "
383 "a rolling restart drains worse, because uvicorn waits for in-flight "
384 "requests while gunicorn SIGKILLs the worker at 30s, dropping every "
385 "websocket that worker still holds. The count is per process, not per "
386 "pod: a pod running N workers holds up to N times this number. 0 or "
387 "negative means no cap.",
388 )
389 LEADER_LOCK_FILE_PATH = confi.str(
390 "LEADER_LOCK_FILE_PATH",
391 "/tmp/opal_server_leader.lock",
392 description="The path to the leader lock file",
393 )
394 POLICY_BUNDLE_SERVER_TYPE = confi.enum(
395 "POLICY_BUNDLE_SERVER_TYPE",
396 PolicyBundleServerType,
397 PolicyBundleServerType.HTTP,
398 description="The type of bundle server e.g. basic HTTP , AWS S3. (affects how we authenticate with it)",
399 )
400 POLICY_BUNDLE_SERVER_TOKEN = confi.str(
401 "POLICY_BUNDLE_SERVER_TOKEN",
402 None,
403 description="Secret token to be sent to API bundle server",
404 )
405 POLICY_BUNDLE_SERVER_TOKEN_ID = confi.str(
406 "POLICY_BUNDLE_SERVER_TOKEN_ID",
407 None,
408 description="The id of the secret token to be sent to API bundle server",
409 )
410 POLICY_BUNDLE_SERVER_AWS_REGION = confi.str(
411 "POLICY_BUNDLE_SERVER_AWS_REGION",
412 "us-east-1",
413 description="The AWS region of the S3 bucket",
414 )
415 POLICY_BUNDLE_TMP_PATH = confi.str(
416 "POLICY_BUNDLE_TMP_PATH",
417 "/tmp/bundle.tar.gz",
418 description="Path for temp policy file, need to be writeable",
419 )
420 POLICY_BUNDLE_GIT_ADD_PATTERN = confi.str(
421 "POLICY_BUNDLE_GIT_ADD_PATTERN",
422 "*",
423 description="File pattern to add files to git default to all the files (*)",
424 )
426 REPO_WATCHER_ENABLED = confi.bool(
427 "REPO_WATCHER_ENABLED",
428 True,
429 description="Enable the repository watcher. In scopes mode (SCOPES=true) this "
430 "same flag enables the scopes sync task on the leader worker: the periodic "
431 "sync_scopes pass, the post-leadership sync of all scopes and the fleet "
432 "purger. It does NOT gate the pre-fork preload, which still clones/fetches "
433 "every registered scope at boot. With the flag off, scopes are registered, "
434 "preloaded at boot and served, but never re-synced or purged afterwards.",
435 )
437 # publisher
438 PUBLISHER_ENABLED = confi.bool(
439 "PUBLISHER_ENABLED", True, description="Enable the publisher"
440 )
442 # broadcaster keepalive
443 BROADCAST_KEEPALIVE_INTERVAL = confi.int(
444 "BROADCAST_KEEPALIVE_INTERVAL",
445 3600,
446 description="the time to wait between sending two consecutive broadcaster keepalive messages",
447 )
448 BROADCAST_KEEPALIVE_TOPIC = confi.str(
449 "BROADCAST_KEEPALIVE_TOPIC",
450 "__broadcast_session_keepalive__",
451 description="the topic on which we should send broadcaster keepalive messages",
452 )
454 # statistics
455 MAX_CHANNELS_PER_CLIENT = confi.int(
456 "MAX_CHANNELS_PER_CLIENT",
457 15,
458 description="max number of records per client, after this number it will not be added to statistics, relevant only if STATISTICS_ENABLED",
459 )
460 STATISTICS_WAKEUP_CHANNEL = confi.str(
461 "STATISTICS_WAKEUP_CHANNEL",
462 "__opal_stats_wakeup",
463 description="The topic a waking-up OPAL server uses to notify others he needs their statistics data",
464 )
465 STATISTICS_STATE_SYNC_CHANNEL = confi.str(
466 "STATISTICS_STATE_SYNC_CHANNEL",
467 "__opal_stats_state_sync",
468 description="The topic other servers with statistics provide their state to a waking-up server",
469 )
470 STATISTICS_SERVER_KEEPALIVE_CHANNEL = confi.str(
471 "STATISTICS_SERVER_KEEPALIVE_CHANNEL",
472 "__opal_stats_server_keepalive",
473 description="The topic workers use to signal they exist and are alive",
474 )
475 STATISTICS_SERVER_KEEPALIVE_TIMEOUT = confi.str(
476 "STATISTICS_SERVER_KEEPALIVE_TIMEOUT",
477 20,
478 description="Timeout for forgetting a server from which a keep-alive haven't been seen (keep-alive frequency would be half of this value)",
479 )
480 SCOPES_PURGE_CHANNEL = confi.str(
481 "SCOPES_PURGE_CHANNEL",
482 "__opal_scope_purge__",
483 description="Pub/sub channel (worker-to-worker, over the broadcaster) used to "
484 "purge GitPolicyFetcher caches fleet-wide when a scope is deleted or "
485 "repointed to a new source. Every worker subscribes and drops its own "
486 "cache entries; the leader is the only actor that may authorize that, "
487 "because only it can check whether a surviving scope still shares the "
488 "source. It does not remove the clone dir.",
489 )
491 # Data updates
492 ALL_DATA_TOPIC = confi.str(
493 "ALL_DATA_TOPIC",
494 DEFAULT_DATA_TOPIC,
495 description="Top level topic for data",
496 )
497 ALL_DATA_ROUTE = confi.str(
498 "ALL_DATA_ROUTE", "/policy-data", description="The route for all policy data"
499 )
500 ALL_DATA_URL = confi.str(
501 "ALL_DATA_URL",
502 confi.delay("http://localhost:7002{ALL_DATA_ROUTE}"),
503 description="URL for all data config [If you choose to have it all at one place]",
504 )
505 DATA_CONFIG_ROUTE = confi.str(
506 "DATA_CONFIG_ROUTE",
507 "/data/config",
508 description="URL to fetch the full basic configuration of data",
509 )
510 DATA_CALLBACK_DEFAULT_ROUTE = confi.str(
511 "DATA_CALLBACK_DEFAULT_ROUTE",
512 "/data/callback_report",
513 description="Exists as a sane default in case the user did not set OPAL_DEFAULT_UPDATE_CALLBACKS",
514 )
516 DATA_CONFIG_SOURCES = confi.model(
517 "DATA_CONFIG_SOURCES",
518 ServerDataSourceConfig,
519 confi.delay(
520 lambda ALL_DATA_URL="", ALL_DATA_TOPIC="": {
521 "config": {
522 "entries": [{"url": ALL_DATA_URL, "topics": [ALL_DATA_TOPIC]}]
523 }
524 }
525 ),
526 description="Configuration of data sources by topics",
527 )
529 DATA_UPDATE_TRIGGER_ROUTE = confi.str(
530 "DATA_CONFIG_ROUTE",
531 "/data/update",
532 description="URL to trigger data update events",
533 )
535 # Git service webhook (Default is Github)
536 POLICY_REPO_WEBHOOK_SECRET = confi.str(
537 "POLICY_REPO_WEBHOOK_SECRET",
538 None,
539 description="The secret for the policy repository webhook",
540 )
541 # The topic the event of the webhook will publish
542 POLICY_REPO_WEBHOOK_TOPIC = "webhook"
543 # Should we check the incoming webhook mentions the branch by name- and not just in the URL
544 POLICY_REPO_WEBHOOK_ENFORCE_BRANCH: bool = confi.bool(
545 "POLICY_REPO_WEBHOOK_ENFORCE_BRANCH",
546 False,
547 description="Enforce branch name in incoming webhook",
548 )
549 # Parameters controlling how the incoming webhook should be read and processed
550 POLICY_REPO_WEBHOOK_PARAMS: GitWebhookRequestParams = confi.model(
551 "POLICY_REPO_WEBHOOK_PARAMS",
552 GitWebhookRequestParams,
553 {
554 "secret_header_name": "x-hub-signature-256",
555 "secret_type": "signature",
556 "secret_parsing_regex": "sha256=(.*)",
557 "event_header_name": "X-GitHub-Event",
558 "event_request_key": None,
559 "push_event_value": "push",
560 },
561 description="Parameters for processing the incoming webhook",
562 )
564 POLICY_REPO_POLLING_INTERVAL = confi.int(
565 "POLICY_REPO_POLLING_INTERVAL",
566 0,
567 description="The polling interval for the policy repository",
568 )
570 ALLOWED_ORIGINS = confi.list(
571 "ALLOWED_ORIGINS", ["*"], description="List of allowed origins for CORS"
572 )
573 FILTER_FILE_EXTENSIONS = confi.list(
574 "FILTER_FILE_EXTENSIONS",
575 [".rego", ".json"],
576 description="List of file extensions to filter. Example: ['.rego', '.json']",
577 )
578 BUNDLE_IGNORE = confi.list(
579 "BUNDLE_IGNORE", [], description="List of patterns to ignore in the bundle"
580 )
582 NO_RPC_LOGS = confi.bool("NO_RPC_LOGS", True, description="Disable RPC logs")
584 # client-api server
585 SERVER_WORKER_COUNT = confi.int(
586 "SERVER_WORKER_COUNT",
587 None,
588 description="(if run via CLI) Worker count for the server [Default calculated to CPU-cores]",
589 )
591 SERVER_HOST = confi.str(
592 "SERVER_HOST",
593 "127.0.0.1",
594 description="(if run via CLI) Address for the server to bind",
595 )
597 SERVER_PORT = confi.str(
598 "SERVER_PORT",
599 None,
600 # Users have experienced errors when kubernetes sets the env-var OPAL_SERVER_PORT="tcp://..." (which fails to parse as a port integer).
601 description="Deprecated, use SERVER_BIND_PORT instead",
602 )
604 SERVER_BIND_PORT = confi.int(
605 "SERVER_BIND_PORT",
606 7002,
607 description="(if run via CLI) Port for the server to bind",
608 )
610 # optional APM tracing with datadog
611 ENABLE_DATADOG_APM = confi.bool(
612 "ENABLE_DATADOG_APM",
613 False,
614 description="Set if OPAL server should enable tracing with datadog APM",
615 )
617 DEBUG_INTERNAL_STATS = confi.bool(
618 "DEBUG_INTERNAL_STATS",
619 False,
620 description=(
621 "Expose GET /internal/git-fetcher-cache-stats with in-memory cache "
622 "sizes and process RSS. For diagnostics/tests only; keep off in production."
623 ),
624 )
626 SCOPES = confi.bool("SCOPES", default=False, description="Enable scopes")
628 SCOPES_REPO_CLONES_SHARDS = confi.int(
629 "SCOPES_REPO_CLONES_SHARDS",
630 1,
631 description="The max number of local clones to use for the same repo (reused across scopes)",
632 )
634 REDIS_URL = confi.str(
635 "REDIS_URL",
636 default="redis://localhost",
637 description="The URL for the Redis server",
638 )
640 BASE_DIR = confi.str(
641 "BASE_DIR",
642 default=pathlib.Path.home() / ".local/state/opal",
643 description="The base directory for OPAL",
644 )
646 POLICY_REFRESH_INTERVAL = confi.int(
647 "POLICY_REFRESH_INTERVAL",
648 default=0,
649 description="Policy polling refresh interval",
650 )
652 SCOPES_STORE_READ_TIMEOUT = confi.float(
653 "SCOPES_STORE_READ_TIMEOUT",
654 10.0,
655 description="Timeout for a scope-store read taken while holding a source's "
656 "lock — the sibling check a delete/repoint purge runs before authorizing "
657 "the fleet-wide cache purge. The Redis client is built without a socket timeout, so without "
658 "this an unreachable store would pin that lock for the life of the process "
659 "and block every later sync, purge and delete for the source. On expiry "
660 "the sibling check fails open, and what that decides is the fleet-wide "
661 "MEMORY purge only — the leader no longer touches the clone tree: a "
662 "DELETE confirms it defensively (its record is already gone, so "
663 "withholding would strand the fleet's cache entries, while over-purging "
664 "self-heals on the surviving sibling's next sync), a REPOINT withholds "
665 "it (the old source's record is still live, just moved). The clone dir "
666 "is unaffected either way: on a store fault the delete floor KEEPS this "
667 "worker's clone rather than risk deleting one a live sibling shares, "
668 "leaving an orphan tracked by PER-15612 "
669 "(0 or negative means no timeout).",
670 )
672 def on_load(self):
673 if self.SERVER_PORT is not None and self.SERVER_PORT.isdigit(): 673 ↛ 675line 673 didn't jump to line 675 because the condition on line 673 was never true
674 # Backward compatibility - if SERVER_PORT is set with a valid value, use it as SERVER_BIND_PORT
675 self.SERVER_BIND_PORT = int(self.SERVER_PORT)
678opal_server_config = OpalServerConfig(prefix="OPAL_")