1# -- Global configuration
3 # -- Override the image registry for all images in the chart
4 # Useful for air-gapped clusters that mirror images to an internal registry
6 # -- Override image pull secrets globally
8 # -- Kubernetes cluster DNS domain
9 clusterDomain: "cluster.local"
10# -- Enable production guard rails: template rendering fails if insecure
11# chart defaults (e.g. the default database password) would reach the
12# cluster. Enabled in values-prod.yaml; leave false for dev/kind.
14# -- Number of KATES replicas
16# -- Container image configuration
19 repository: chainreg.biz/chainguard-private/kates-fips
20 # -- Image tag. Must be the release that carries the non-blocking health
21 # endpoints the probe defaults below depend on (>= 1.22.0).
22 tag: 1.24.0-r1@sha256:c690097b5bc51234a5e3184847038e6bb18ca283d47815a1da1af5022bf93446
23 # -- Image pull policy
24 pullPolicy: IfNotPresent
25# -- Secrets for pulling images from private registries
27# -- Override the chart name
29# -- Override the full release name
31# -- DNS policy for pods (ClusterFirst, Default, None, ClusterFirstWithHostNet)
33# -- Custom DNS configuration (nameservers, searches, options)
35# -- Service account configuration
37 # -- Create a ServiceAccount
39 # -- ServiceAccount name (auto-generated if empty)
41 # -- Extra annotations for the ServiceAccount
43 # -- Automount the service account token
45# -- RBAC configuration
47 # -- Create ClusterRole and ClusterRoleBinding
49 # -- Grant the writes the direct Kubernetes chaos backend makes itself
50 # (KATES_CHAOS_PROVIDER=kubernetes, or hybrid without Litmus): NetworkPolicies,
51 # StatefulSet scaling, and pod ephemeral containers. Litmus runs its
52 # experiments as litmus-admin and needs none of this.
54 # -- Extra RBAC rules (e.g. for Litmus Chaos integration)
56# -- Extra annotations to add to pods
58# -- Extra labels to add to pods
60# -- Kubernetes Service configuration
64 # -- HTTP service port
66 # -- NodePort value (only when type=NodePort)
68 # -- Application protocol hint for service mesh
70 # -- Extra annotations for the Service
72# -- gRPC port configuration
74 # -- gRPC container/service port
76 # -- gRPC NodePort value (only when type=NodePort)
78# -- Ingress configuration
82 # -- Ingress class name (e.g. nginx)
84 # -- Extra Ingress annotations
86 # -- cert-manager integration
88 # -- Enable cert-manager annotations
90 # -- Issuer or ClusterIssuer name
92 # -- Issuer kind (ClusterIssuer or Issuer)
93 issuerKind: ClusterIssuer
100 # -- TLS configuration
102 # -- gRPC Ingress (separate Ingress resource for gRPC traffic)
104 # -- Enable gRPC Ingress
106 # -- gRPC Ingress hostname
107 host: grpc.kates.local
108 # -- Extra gRPC Ingress annotations
110 # -- gRPC TLS configuration
112# -- Resource requests and limits for the KATES container
120# -- Flyway schema migration behaviour
122 # -- Repair the schema history on startup before migrating.
124 # DEVELOPMENT ONLY. It fixes "Migration checksum mismatch", which happens when
125 # an unreleased migration is edited after a local database already applied it.
126 # On a real database that mismatch means the file and the schema genuinely
127 # disagree, and silently rewriting the history hides it — so this stays off by
128 # default and is enabled in the local overlays.
130# -- Container args, replacing the image's CMD. Empty means "use the image's
131# own default". The native image sets its heap percentage through CMD, since a
132# native binary ignores JAVA_TOOL_OPTIONS; override it here rather than
133# rebuilding the image.
135# -- JVM configuration
137# JVM IMAGE ONLY. A GraalVM native image does not read JAVA_TOOL_OPTIONS, so
138# everything here is silently ignored when running a `-native` tag — see
139# values-native.yaml, which empties it for that reason.
141 # -- JVM options passed via JAVA_TOOL_OPTIONS (ignored by the native image)
142 options: "-Xms512m -Xmx2560m -XX:+UseZGC -XX:+ZGenerational"
143# -- Pod-level security context (PSS restricted compliant)
144# NOTE: runAsUser and fsGroup are intentionally unset so that platforms
145# like OpenShift can assign random UIDs via their Security Context Constraints.
146# Set them explicitly in your environment values file if needed.
153# -- Container-level security context (PSS restricted compliant)
154containerSecurityContext:
155 allowPrivilegeEscalation: false
156 readOnlyRootFilesystem: true
161# -- Grace period for pod termination (seconds)
162terminationGracePeriodSeconds: 60
163# -- Deployment update strategy
169# -- Horizontal Pod Autoscaler configuration
173 # -- Minimum replicas
175 # -- Maximum replicas
177 # -- Target CPU utilization percentage
178 targetCPUUtilizationPercentage: 80
179 # -- Target memory utilization percentage (empty to disable)
180 targetMemoryUtilizationPercentage: ""
181 # -- Custom metrics for HPA
183 # -- Scale behavior policies
186 stabilizationWindowSeconds: 300
192 stabilizationWindowSeconds: 30
197# -- Pod Disruption Budget configuration
201 # -- Minimum available pods (overrides maxUnavailable)
203 # -- Maximum unavailable pods
205 # -- Unhealthy pod eviction policy
206 unhealthyPodEvictionPolicy: IfHealthy
207# -- Priority class name for pod scheduling
209# -- Topology spread constraints for pod scheduling
210topologySpreadConstraints: []
211# -- Node selector for pod scheduling
213# -- Tolerations for pod scheduling
215# -- Affinity rules (includes default pod anti-affinity for HA)
218 preferredDuringSchedulingIgnoredDuringExecution:
223 - key: app.kubernetes.io/name
227 topologyKey: kubernetes.io/hostname
228# -- Extra environment variables for the KATES container
230# -- Extra init containers (e.g. for Vault secret fetching, schema migration)
231extraInitContainers: []
234# -- Extra volume mounts
236# -- API key authentication
238 # -- Enable API key authentication
240 # -- API key value (auto-generated 32-char key if empty)
242 # -- Use an existing Secret instead of creating one
244 # -- Key in the existing Secret
246# -- Health probe configuration
248# Every probe points at a MicroProfile health endpoint, never at a business
249# endpoint. `/api/health` looks like the obvious readiness URL but calls
250# ClusterHealthService.isReachable() — an uncached AdminClient round-trip with a
251# 5s blocking get — on every request. Since a probe's timeoutSeconds defaults to
252# 1s, a cluster whose Kafka is still coming up fails EVERY readiness probe and
253# the pod never goes Ready, even though the app is fine. `/q/health/ready` reads
254# the periodically-refreshed KafkaReachabilityCache instead and never blocks on a
257 # -- Startup probe (gates readiness/liveness probes)
259 path: /q/health/started
260 # 60 x 2s = 120s of JVM boot budget. 30 (60s) is not enough on a
261 # CPU-constrained node: the kubelet kills the container mid-boot and the
262 # restart loop reads as "kates takes forever to start".
268 path: /q/health/ready
269 initialDelaySeconds: 5
276 initialDelaySeconds: 15
280# -- Container lifecycle hooks
282 # -- Seconds to sleep in preStop hook (allows LB drain)
283 preStopSleepSeconds: 5
284# -- Init container configuration
286 # -- Busybox image for wait-for-postgres init container
288 image: chainreg.biz/chainguard-private/netcat:1.238-r1@sha256:b961470bd7a9a1ece5f8bf51b2accd244cebc7a83767ab08da80821e403eb140
289 pullPolicy: IfNotPresent
290# -- Kafka connection configuration
292 # -- Kafka bootstrap servers
293 bootstrapServers: "krafter-kafka-bootstrap.kafka.svc:9092"
294 # -- Namespace where CDC integration test KafkaTopic CRs are managed (empty = auto-detect)
296 # -- Client rack ID for rack-aware consumption
298 # -- Security protocol
300 protocol: SASL_PLAINTEXT
301 # -- SASL authentication
306 mechanism: SCRAM-SHA-512
308 username: kates-backend
309 # -- K8s Secret name containing the SASL password
310 secretName: kates-backend
311 # -- Key in the Secret containing the SASL password
313 # -- OAuthBearer configuration (when mechanism=OAUTHBEARER)
318 clientSecretKey: client-secret
319 # -- SSL/TLS configuration
323 # -- Truststore configuration
328 # -- Keystore configuration
333# -- Trogdor coordinator configuration
335 # -- Trogdor coordinator URL
336 coordinatorUrl: "http://trogdor-coordinator.kafka.svc:8889"
337 # -- Trogdor agents that run tasks (KATES_TROGDOR_AGENT_NODES): node names
338 # from the coordinator's platform config, comma-separated; a run's tasks take
339 # them in turn. node0 is the name in the trogdor.conf Kafka ships. A name the
340 # coordinator does not know fails the task with "Unknown node names".
342# -- Prometheus the backend queries for a disruption's Kafka metrics
344 # -- Prometheus base URL (KATES_PROMETHEUS_URL). The default is the Service
345 # kube-prometheus-stack creates for charts/monitoring installed as release
346 # `monitoring` in namespace `monitoring`, which is what `kates deploy` does
347 # (it sets this for the namespace it uses). `make monitoring` installs that
348 # release in namespace `kafka`: set
349 # http://monitoring-kube-prometheus-prometheus.kafka.svc:9090 there, and
350 # networkPolicy.prometheus.namespace to kafka. Unreachable, a disruption's
351 # Prometheus metrics are missing and its SLA verdict lists them unevaluated.
352 url: "http://monitoring-kube-prometheus-prometheus.monitoring.svc:9090"
353# -- Benchmark engine configuration
355 # -- Default benchmark backend (native or trogdor)
356 defaultBackend: "native"
357# -- Fault tolerance timeout configuration (ms)
360 # -- Timeout for all TopicService operations
362 clusterHealthService:
363 # -- Default timeout for cluster health operations
364 defaultTimeoutMs: "35000"
365 # -- Timeout for reachability check
366 reachableTimeoutMs: "10000"
367 # -- Timeout for full health check
368 healthCheckTimeoutMs: "60000"
369 consumerGroupService:
370 # -- Timeout for consumer group operations
372# -- NetworkPolicy configuration
374 # -- Enable NetworkPolicy
379 - namespaceSelector: {}
384 - namespaceSelector: {}
388 # -- Kafka egress configuration
392 # -- Prometheus egress configuration: the namespace prometheus.url points
393 # into. The policy used to allow `monitoring` whatever this said.
395 namespace: monitoring
397 # -- Extra egress rules
399# -- Observability and monitoring
401 # -- Prometheus ServiceMonitor
403 # -- Enable ServiceMonitor
405 # -- Target namespace for the ServiceMonitor
411 # -- Extra labels for ServiceMonitor
413 # -- Metric relabeling rules
414 metricRelabelings: []
415 # -- Relabeling rules
417 # -- PrometheusRule for alerting
419 # -- Enable PrometheusRule
421 # -- Target namespace for the PrometheusRule
428 expr: up{job="{{ include \"kates.fullname\" . }}"} == 0
433 summary: "KATES instance is down"
434 description: "KATES pod {{ $labels.pod }} has been down for more than 5 minutes."
435 - alert: KatesHighErrorRate
436 expr: rate(http_server_requests_seconds_count{job="{{ include \"kates.fullname\" . }}", status=~"5.."}[5m]) > 0.1
441 summary: "KATES high error rate"
442 description: "KATES is returning more than 10% 5xx responses."
443 # -- GONE in 0.9.0: the "KATES — Overview" board (dashboards/kates-overview/)
444 # is delivered by charts/monitoring with every other board in dashboards/
445 # (its `dashboards.enabled`), scoped to a release by its `$job` template
446 # variable. A release that still sets any `metrics.grafanaDashboard.*` key
447 # is REFUSED with the new location named (this chart has no values schema,
448 # so without the refusal the key would be accepted and silently ignored).
449# -- Bundled PostgreSQL configuration.
450# A self-contained single-instance StatefulSet on the official `postgres`
451# image (no Bitnami). For production HA, set enabled=false and point
452# externalDatabase at a managed cluster (CloudNativePG recommended).
454 # -- Deploy bundled PostgreSQL
456 # -- PostgreSQL image (official library image; Alpine variants also work)
458 repository: chainreg.biz/chainguard-private/postgres-fips
459 tag: 16.15-r5@sha256:720d154c20c41279f2872ee0e158fc60d9a11fecd6d4ac793f3e1e49b0022f08
460 # -- Database authentication
464 # -- Database username
466 # -- Database password (ignored if existingSecret is set)
468 # -- Use an existing Secret for DB credentials
470 # -- Key in the existing Secret
471 passwordKey: "password"
472 # -- Node selector for the embedded PostgreSQL pod (dev/test only)
474 # -- Tolerations for the embedded PostgreSQL pod
476 # -- Affinity rules for the embedded PostgreSQL pod
478 # -- PVC storage configuration
482 # -- Storage class (empty for default)
484 # -- PostgreSQL resource requests and limits
492 # -- PostgreSQL pod security context
497 # -- PostgreSQL container security context
498 containerSecurityContext:
499 allowPrivilegeEscalation: false
503 # -- JDBC connection pool settings
505 # -- Maximum pool size
507 # -- Minimum pool size
509# -- External database configuration (when postgresql.enabled=false).
510# Recommended for production HA: run a CloudNativePG `Cluster` (official
511# postgres images, streaming replication, backups) and point host/port at its
512# `-rw` service, e.g. host: kates-db-rw.database.svc — no Bitnami involved.
514 # -- Enable external database
522 # -- Database username
524 # -- Database password (ignored if existingSecret is set)
526 # -- Use an existing Secret for DB credentials
528 # -- Key in the existing Secret
529 passwordKey: "password"
530# -- Default benchmark parameters (fallback for all test types)
532 replicationFactor: "3"
534 minInsyncReplicas: "2"
538 compressionType: "lz4"
540 numRecords: "1000000"
545# -- Per-test-type parameter overrides
547 # -- LOAD: baseline throughput test
552 numRecords: "1000000"
555 # -- STRESS: high-concurrency, large batches
560 numRecords: "5000000"
563 # -- SPIKE: burst traffic, low-latency
568 compressionType: "none"
569 numRecords: "2000000"
571 # -- ENDURANCE: long-running, rate-limited
573 numRecords: "10000000"
575 durationMs: "3600000"
576 # -- VOLUME: large records, high batch size
582 numRecords: "2000000"
583 # -- CAPACITY: max parallelism
588 numRecords: "10000000"
589 durationMs: "1200000"
591 # -- ROUNDTRIP: latency-focused
595 compressionType: "none"
598# -- PostgreSQL backup CronJob
600 # -- Enable backup CronJob
602 # -- Cron schedule expression
603 schedule: "0 2 * * *"
604 # -- PostgreSQL image for pg_dump
605 image: chainreg.biz/chainguard-private/postgres-fips:16.15-r5@sha256:720d154c20c41279f2872ee0e158fc60d9a11fecd6d4ac793f3e1e49b0022f08
606 # -- Days to retain backups
608 # -- Backup persistence
614 # -- Backup job resources
622# -- Pre-upgrade Flyway migration Job
624 # -- Enable migration Job (runs as pre-upgrade Helm hook)
626 # -- Override image (defaults to KATES app image)
627 image: chainreg.biz/chainguard-private/kates-fips:1.24.0-r1@sha256:c690097b5bc51234a5e3184847038e6bb18ca283d47815a1da1af5022bf93446
628 # -- Custom command (overrides default Flyway migration)
630 # -- Maximum retry attempts
632 # -- Maximum Job duration (seconds)
633 activeDeadlineSeconds: 600
634 # -- Migration job resources
642# -- Stale test run cleanup CronJob
644 # -- Enable cleanup CronJob
646 # -- Cron schedule expression
647 schedule: "0 4 * * 0"
648 # -- Delete completed test runs older than N days
650 # -- Curl image for API calls
651 image: chainreg.biz/chainguard-private/min-toolkit-debug-fips:latest@sha256:76f732b60df11113cc47422ea95d65543229da16b6e23f5f8d551f8bb0c3ccce
652# -- Kyverno pod security policies (requires Kyverno controller in cluster)
654 # -- Enable Kyverno ClusterPolicy for the kates namespace
656 # -- Policy action: Audit (observe only) or Enforce (block non-compliant)
658 # -- Enable mutation rules to auto-inject security contexts
660 # -- Enable image registry restriction
661 restrictRegistries: false
662 # -- Allowed image registries (when restrictRegistries is true)
669 # -- GONE in 0.9.0: the Kyverno board (dashboards/kyverno-security/) is
670 # delivered by charts/monitoring with every other board in dashboards/ (its
671 # `dashboards.enabled`) — one Kyverno per cluster, one board, and it belongs
672 # where the rest of the cluster's boards live. `grafanaDashboard` (both the
673 # bool and the map shape) and `grafanaDashboardNamespace` are REFUSED if
674 # set, with the new location named.
675 # -- Cosign image signature verification
677 # -- Enable image signature verification via Cosign
679 # -- Image patterns requiring valid signatures
681 - "ghcr.io/bmscomp/*"
682 # -- Cosign public key (PEM format) — set via --set-file or Secrets
684 # -- Auto-generate default-deny NetworkPolicies for new namespaces
685 networkPolicyGeneration:
687 # -- Namespaces to exclude from default-deny generation
699 # -- PolicyExceptions for dev/test namespaces
701 # -- Enable PolicyException resources
703 # -- Dev namespaces where policies are relaxed
707 # -- Policies to create exceptions for
709 - kates-pod-security-standards
710 # -- Rules to exempt in dev namespaces
712 - validate-readonly-rootfs
713 - require-resource-limits
714 - disallow-latest-tag
715# -- Monitoring integration
717 # -- Enable telemetry export (Jaeger)