2# commonLabels -- set of labels that will be applied to all the resources for the operator
4# commonAnnotations -- set of annotations that will be applied to all the resources for the operator
7 # look details in `kubectl explain deployment.spec.strategy`
11 # crdHook.enabled -- enable automatic CRD installation/update via pre-install/pre-upgrade hooks
12 # when disabled, CRDs must be installed manually using kubectl apply
15 # crdHook.image.repository -- image repository for CRD installation job
16 repository: chainreg.biz/chainguard-private/kubectl
17 # crdHook.image.tag -- image tag for CRD installation job.
18 # registry.k8s.io/kubectl publishes no `latest` tag, so this must name an explicit version
19 tag: latest@sha256:00df66dcd7275b2445777a83e705e1acee6ff81d82dad1596470af04b6bd16ca
20 # crdHook.image.pullPolicy -- image pull policy for CRD installation job
21 pullPolicy: IfNotPresent
22 # crdHook.imagePullSecrets -- image pull secrets for CRD installation job
23 # possible value format `[{"name":"your-secret-name"}]`,
24 # check `kubectl explain pod.spec.imagePullSecrets` for details
26 # crdHook.resources -- resource limits and requests for CRD installation job
34 # crdHook.nodeSelector -- node selector for CRD installation job
36 # crdHook.tolerations -- tolerations for CRD installation job
38 # crdHook.affinity -- affinity for CRD installation job
40 # crdHook.annotations -- additional annotations for CRD installation job
42 # crdHook.podAnnotations -- additional annotations for CRD installation job pod template
43 # useful to opt out of service mesh injection, e.g. `sidecar.istio.io/inject: "false"`
45 # crdHook.podSecurityContext -- pod-level security context for CRD installation job
46 # required by some admission policies (e.g. Kyverno `restrict-seccomp-strict`)
47 # check `kubectl explain pod.spec.securityContext` for details
48 podSecurityContext: {}
52 # type: RuntimeDefault
53 # crdHook.containerSecurityContext -- container security context for CRD installation job
54 # check `kubectl explain pod.spec.containers.securityContext` for details
55 containerSecurityContext: {}
56 # allowPrivilegeEscalation: false
62 # type: RuntimeDefault
65 # operator.image.registry -- optional image registry prefix (e.g. 1234567890.dkr.ecr.us-east-1.amazonaws.com)
67 # operator.image.repository -- image repository
68 repository: chainreg.biz/chainguard-private/clickhouse-operator
69 # operator.image.tag -- image tag (chart's appVersion value will be used if not set)
70 tag: 0.27.4-r0@sha256:3ecc08fbe9c556e33e94b8eb1e9f4386aa3b72552536958537592081ebf62c17
71 # operator.image.pullPolicy -- image pull policy
72 pullPolicy: IfNotPresent
73 containerSecurityContext: {}
74 # operator.resources -- custom resource configuration, check `kubectl explain pod.spec.containers.resources` for details
83 # operator.priorityClassName -- priority class name for the clickhouse-operator deployment, check `kubectl explain pod.spec.priorityClassName` for details
86 # operator.env -- additional environment variables for the clickhouse-operator container in deployment
87 # possible format value `[{"name": "SAMPLE", "value": "text"}]`
89 # operator.livenessProbe -- optional liveness probe for the clickhouse-operator container
90 # check `kubectl explain pod.spec.containers.livenessProbe` for details
95 # initialDelaySeconds: 10
98 # operator.readinessProbe -- optional readiness probe for the clickhouse-operator container
99 # check `kubectl explain pod.spec.containers.readinessProbe` for details
104 # initialDelaySeconds: 5
110 # metrics.image.registry -- optional image registry prefix (e.g. 1234567890.dkr.ecr.us-east-1.amazonaws.com)
112 # metrics.image.repository -- image repository
113 repository: chainreg.biz/chainguard-private/clickhouse-operator-metrics-exporter
114 # metrics.image.tag -- image tag (chart's appVersion value will be used if not set)
115 tag: 0.27.4-r0@sha256:d1aa5ce47cdb5db2ae864866a42f2965c7f3864124e2edb71477febb45e9f269
116 # metrics.image.pullPolicy -- image pull policy
117 pullPolicy: IfNotPresent
118 containerSecurityContext: {}
119 # metrics.resources -- custom resource configuration
128 # metrics.env -- additional environment variables for the deployment of metrics-exporter containers
129 # possible format value `[{"name": "SAMPLE", "value": "text"}]`
131 # metrics.livenessProbe -- optional liveness probe for the metrics-exporter container
132 # check `kubectl explain pod.spec.containers.livenessProbe` for details
137 # initialDelaySeconds: 10
140 # metrics.readinessProbe -- optional readiness probe for the metrics-exporter container
141 # check `kubectl explain pod.spec.containers.readinessProbe` for details
146 # initialDelaySeconds: 5
149# imagePullSecrets -- image pull secret for private images in clickhouse-operator pod
150# possible value format `[{"name":"your-secret-name"}]`,
151# check `kubectl explain pod.spec.imagePullSecrets` for details
153# podLabels -- labels to add to the clickhouse-operator pod
155# podAnnotations -- annotations to add to the clickhouse-operator pod, check `kubectl explain pod.spec.annotations` for details
156# @default -- check the `values.yaml` file
158 prometheus.io/port: '8888'
159 prometheus.io/scrape: 'true'
160 clickhouse-operator-metrics/port: '9999'
161 clickhouse-operator-metrics/scrape: 'true'
162# watchNamespaces -- namespaces where the operator watches for ClickHouseInstallation resources.
163# Sets config.yaml watch.namespaces.include (the exclude list is not exposed here). If empty, the
164# operator watches only its own namespace (or all namespaces when running in kube-system).
165# Use [".*"] to watch all namespaces. Entries are regexps and are Helm-templated, so avoid a literal
166# "{{" in a namespace/regexp.
167# Example: watchNamespaces: ["clickhouse", "my-other-namespace"]
169# nameOverride -- override name of the chart
171# fullnameOverride -- full name of the chart.
174 # serviceAccount.create -- specifies whether a service account should be created
176 # serviceAccount.annotations -- annotations to add to the service account
178 # serviceAccount.name -- the name of the service account to use; if not set and create is true, a name is generated using the fullname template
181 # rbac.create -- specifies whether rbac resources should be created
183 # rbac.namespaceScoped -- specifies whether to create roles and rolebindings at the cluster level or namespace level
184 namespaceScoped: false
186 # secret.create -- create a secret with operator credentials
188 # secret.username -- operator credentials username
189 username: clickhouse_operator
190 # secret.password -- operator credentials password
191 password: clickhouse_operator_password
192# nodeSelector -- node for scheduler pod assignment, check `kubectl explain pod.spec.nodeSelector` for details
194# tolerations -- tolerations for scheduler pod assignment, check `kubectl explain pod.spec.tolerations` for details
196# affinity -- affinity for scheduler pod assignment, check `kubectl explain pod.spec.affinity` for details
198# podSecurityContext - operator deployment SecurityContext, check `kubectl explain pod.spec.securityContext` for details
199podSecurityContext: {}
200# topologySpreadConstraints - topologySpreadConstraints affinity for scheduler pod assignment, check `kubectl explain pod.spec.topologySpreadConstraints` for details
201topologySpreadConstraints: []
203 # serviceMonitor.enabled -- ServiceMonitor Custom resource is created for a [prometheus-operator](https://github.com/prometheus-operator/prometheus-operator)
204 # In serviceMonitor will be created two endpoints ch-metrics on port 8888 and op-metrics # 9999, plus a separate keeper-metrics ServiceMonitor when `serviceMonitor.keeperMetrics.enabled` is also set. You can specify interval, scrapeTimeout, relabelings, metricRelabelings for each endpoint below
206 # serviceMonitor.additionalLabels -- additional labels for service monitor
209 # serviceMonitor.interval for ch-metrics endpoint --
211 # serviceMonitor.scrapeTimeout for ch-metrics endpoint -- Prometheus ServiceMonitor scrapeTimeout. If empty, Prometheus uses the global scrape timeout unless it is less than the target's scrape interval value in which the latter is used.
213 # serviceMonitor.relabelings for ch-metrics endpoint -- Prometheus [RelabelConfigs] to apply to samples before scraping
215 # serviceMonitor.metricRelabelings for ch-metrics endpoint -- Prometheus [MetricRelabelConfigs] to apply to samples before ingestio
216 metricRelabelings: []
218 # serviceMonitor.interval for op-metrics endpoint --
220 # serviceMonitor.scrapeTimeout for op-metrics endpoint -- Prometheus ServiceMonitor scrapeTimeout. If empty, Prometheus uses the global scrape timeout unless it is less than the target's scrape interval value in which the latter is used.
222 # serviceMonitor.relabelings for op-metrics endpoint -- Prometheus [RelabelConfigs] to apply to samples before scraping
224 # serviceMonitor.metricRelabelings for op-metrics endpoint -- Prometheus [MetricRelabelConfigs] to apply to samples before ingestio
225 metricRelabelings: []
227 # serviceMonitor.keeperMetrics.enabled -- create a separate ServiceMonitor scraping the native ClickHouse Keeper prometheus endpoint on the Keeper Services the operator manages. Unlike ClickHouse server metrics, Keeper metrics do not go through the operator's metrics-exporter - Keeper serves Prometheus itself. Off by default because it needs the CHK to opt in first: operator-generated Keeper Services publish only the `zk`/`zk-secure`/`raft` ports, so the CHK must both enable the endpoint via `prometheus/*` settings and name the port in `spec.templates.serviceTemplates` (a serviceTemplate REPLACES the default port list, so re-declare `zk` and `raft` there). Without that the ServiceMonitor matches no targets and reports nothing.
229 # serviceMonitor.keeperMetrics.port -- name of the Keeper Service port exposing the native prometheus endpoint. Must match the port name in the CHK serviceTemplate. `prometheus` follows this repo's own Keeper examples (port 7000); charts that name it `metrics` need this overridden
231 # serviceMonitor.keeperMetrics.selector -- matchLabels selecting the Keeper Services to scrape. The default matches every Keeper Service the operator manages. Note the operator creates several Services per CHK (CR-scope plus per-host peer and client), all carrying this label, so if more than one of them names the metrics port the same pod is scraped more than once - narrow with `clickhouse-keeper.altinity.com/Service: chk` (CR-scope) or `: host` (per-host) to pick a single tier
232 # @default -- check the `values.yaml` file
234 clickhouse-keeper.altinity.com/app: chop
235 # serviceMonitor.keeperMetrics.namespaceSelector -- namespaces where prometheus-operator will discover keeper services, `any: true` means all namespaces. Entries under `matchNames` must be literal namespace names, not the regexps `watchNamespaces` accepts. Cluster-wide discovery also requires the Prometheus service account to hold cluster-scoped RBAC on services/endpoints, and Prometheus to select this ServiceMonitor (see `serviceMonitor.additionalLabels`)
238 # serviceMonitor.keeperMetrics.interval -- Prometheus scrape interval for the keeper-metrics endpoint
240 # serviceMonitor.keeperMetrics.scrapeTimeout -- Prometheus ServiceMonitor scrapeTimeout. If empty, Prometheus uses the global scrape timeout unless it is less than the target's scrape interval value in which the latter is used.
242 # serviceMonitor.keeperMetrics.relabelings -- Prometheus [RelabelConfigs] to apply to samples before scraping
244 # serviceMonitor.keeperMetrics.metricRelabelings -- Prometheus [MetricRelabelConfigs] to apply to samples before ingestion
245 metricRelabelings: []
246# configs -- clickhouse operator configs
247# @default -- check the `values.yaml` file for the config content (auto-generated from latest operator release)
251 01-clickhouse-01-listen.xml: |
253 <!-- This file is auto-generated -->
254 <!-- Do not edit this file - all changes would be lost -->
255 <!-- Edit appropriate template in the following folder: -->
256 <!-- deploy/builder/templates-config -->
259 <!-- Listen wildcard address to allow accepting connections from other containers and host network. -->
260 <listen_host>::</listen_host>
261 <listen_host>0.0.0.0</listen_host>
262 <listen_try>1</listen_try>
264 01-clickhouse-02-logger.xml: |
266 <!-- This file is auto-generated -->
267 <!-- Do not edit this file - all changes would be lost -->
268 <!-- Edit appropriate template in the following folder: -->
269 <!-- deploy/builder/templates-config -->
273 <!-- Possible levels: https://github.com/pocoproject/poco/blob/devel/Foundation/include/Poco/Logger.h#L439 -->
275 <log>/var/log/clickhouse-server/clickhouse-server.log</log>
276 <errorlog>/var/log/clickhouse-server/clickhouse-server.err.log</errorlog>
279 <!-- Default behavior is autodetection (log to console if not daemon mode and is tty) -->
283 01-clickhouse-03-query_log.xml: |
285 <!-- This file is auto-generated -->
286 <!-- Do not edit this file - all changes would be lost -->
287 <!-- Edit appropriate template in the following folder: -->
288 <!-- deploy/builder/templates-config -->
291 <query_log replace="1">
292 <database>system</database>
293 <table>query_log</table>
294 <engine>Engine = MergeTree PARTITION BY event_date ORDER BY event_time TTL event_date + interval 30 day</engine>
295 <flush_interval_milliseconds>7500</flush_interval_milliseconds>
297 <query_thread_log remove="1"/>
299 01-clickhouse-04-part_log.xml: |
301 <!-- This file is auto-generated -->
302 <!-- Do not edit this file - all changes would be lost -->
303 <!-- Edit appropriate template in the following folder: -->
304 <!-- deploy/builder/templates-config -->
307 <part_log replace="1">
308 <database>system</database>
309 <table>part_log</table>
310 <engine>Engine = MergeTree PARTITION BY event_date ORDER BY event_time TTL event_date + interval 30 day</engine>
311 <flush_interval_milliseconds>7500</flush_interval_milliseconds>
314 01-clickhouse-05-trace_log.xml: |-
316 <!-- This file is auto-generated -->
317 <!-- Do not edit this file - all changes would be lost -->
318 <!-- Edit appropriate template in the following folder: -->
319 <!-- deploy/builder/templates-config -->
322 <trace_log replace="1">
323 <database>system</database>
324 <table>trace_log</table>
325 <engine>Engine = MergeTree PARTITION BY event_date ORDER BY event_time TTL event_date + interval 30 day</engine>
326 <flush_interval_milliseconds>7500</flush_interval_milliseconds>
332 # This file is auto-generated
333 # Do not edit this file - all changes would be lost
334 # Edit appropriate template in the following folder:
335 # deploy/builder/templates-config
338 # Template parameters available:
342 # CH_CREDENTIALS_SECRET_NAMESPACE=
343 # CH_CREDENTIALS_SECRET_NAME=clickhouse-operator
346 ################################################
350 ################################################
352 # Namespaces where clickhouse-operator watches for events.
353 # Concurrently running operators should watch on different namespaces.
354 # `include` and `exclude` accept literal namespace names or regexp patterns.
355 # Empty `include` watches the operator's own namespace (or all namespaces when
356 # the operator runs in `kube-system`); use [".*"] to force watch-all elsewhere.
357 # Empty `exclude` matches none. `exclude` is applied after `include`.
361 # Behavior when ClickHouseOperatorConfiguration changes: none | restart
366 ################################################
368 ## Configuration files section
370 ################################################
372 # Each 'path' can be either absolute or relative.
373 # In case path is absolute - it is used as is
374 # In case path is relative - it is relative to the folder where configuration file you are reading right now is located.
376 # Path to the folder where ClickHouse configuration files common for all instances within a CHI are located.
378 # Path to the folder where ClickHouse configuration files unique for each instance (host) within a CHI are located.
380 # Path to the folder where ClickHouse configuration files with users' settings are located.
381 # Files are common for all instances within a CHI.
383 ################################################
385 ## Configuration users section
387 ################################################
389 # Default settings for user accounts, created by the operator.
390 # IMPORTANT. These are not access credentials or settings for 'default' user account,
391 # it is a template for filling out missing fields for all user accounts to be created by the operator,
392 # with the following EXCEPTIONS:
393 # 1. 'default' user account DOES NOT use provided password, but uses all the rest of the fields.
394 # Password for 'default' user account has to be provided explicitly, if to be used.
395 # 2. CHOP user account DOES NOT use:
396 # - profile setting. It uses predefined profile called 'clickhouse_operator'
397 # - quota setting. It uses empty quota name.
398 # - networks IP setting. Operator specifies 'networks/ip' user setting to match operators' pod IP only.
399 # - password setting. Password for CHOP account is used from 'clickhouse.access.*' section
401 # Default values for ClickHouse user account(s) created by the operator
402 # 1. user/profile - string
403 # 2. user/quota - string
404 # 3. user/networks/ip - multiple strings
405 # 4. user/password - string
406 # These values can be overwritten on per-user basis.
413 ################################################
415 ## Configuration network section
417 ################################################
419 # Default host_regexp to limit network connectivity from outside
420 hostRegexpTemplate: "(chi-{chi}-[^.]+\\d+-\\d+|clickhouse\\-{chi})\\.{namespace}\\.svc\\.cluster\\.local$"
421 ################################################
423 ## Configuration restart policy section
424 ## Configuration restart policy describes what configuration changes require ClickHouse restart
426 ################################################
427 configurationRestartPolicy:
430 # Special version of "*" - default version - has to satisfy all ClickHouse versions.
431 # Default version will also be used in case ClickHouse version is unknown.
432 # ClickHouse version may be unknown due to host being down - for example, because of incorrect "settings" section.
433 # ClickHouse is not willing to start in case incorrect/unknown settings are provided in config file.
436 # see https://kb.altinity.com/altinity-kb-setup-and-maintenance/altinity-kb-server-config-files/#server-config-configxml-sections-which-dont-require-restart
437 # to be replaced with "select * from system.server_settings where changeable_without_restart = 'No'"
440 - settings/access_control_path: "no"
441 - settings/dictionaries_config: "no"
442 - settings/max_server_memory_*: "no"
443 - settings/max_*_to_drop: "no"
444 - settings/max_concurrent_queries: "no"
445 - settings/models_config: "no"
446 - settings/user_defined_executable_functions_config: "no"
448 - settings/logger/*: "no"
449 - settings/macros/*: "no"
450 - settings/remote_servers/*: "no"
451 - settings/user_directories/*: "no"
452 # these settings should not lead to pod restarts
453 - settings/display_secrets_in_show_and_select: "no"
456 - files/config.d/*.xml: "yes"
457 - files/config.d/*dict*.xml: "no"
458 - files/config.d/*no_restart*: "no"
459 # exceptions in default profile
460 - profiles/default/background_*_pool_size: "yes"
461 - profiles/default/max_*_for_server: "yes"
464 - settings/logger: "yes"
465 #################################################
467 ## Access to ClickHouse instances
469 ################################################
471 # Possible values for 'scheme' are:
472 # 1. http - force http to be used to connect to ClickHouse instances
473 # 2. https - force https to be used to connect to ClickHouse instances
474 # 3. Auto - either http or https is selected based on open ports
476 # ClickHouse credentials (username, password and port) to be used by the operator to connect to ClickHouse instances.
477 # These credentials are used for:
478 # 1. Metrics requests
479 # 2. Schema maintenance
480 # User with these credentials can be specified in additional ClickHouse .xml config files,
481 # located in 'clickhouse.configuration.file.path.user' folder
484 # Inline PEM CA bundle the operator uses to verify ClickHouse server TLS.
486 # Alternate source for rootCA — a Secret in the operator's namespace.
487 # Mutually exclusive with the inline rootCA above (inline wins). Empty
488 # `name` = not used (no-op). When `key` is empty, the operator tries
489 # "ca.crt" then "tls.crt". Resolved once at config load (an operator
490 # restart picks up a rotated Secret).
494 # Location of the k8s Secret with username and password to be used by the operator to connect to ClickHouse instances.
495 # Can be used instead of explicitly specified username and password available in sections:
496 # - clickhouse.access.username
497 # - clickhouse.access.password
498 # Secret should have two keys:
502 # Empty `namespace` means that k8s secret would be looked in the same namespace where operator's pod is running.
504 # Empty `name` means no k8s Secret would be looked for
505 name: '{{ include "altinity-clickhouse-operator.fullname" . }}'
506 # Port where to connect to ClickHouse instances to
508 # Timeouts used to limit connection and queries from the operator to ClickHouse instances
509 # Specified in seconds.
511 # Timout to setup connection from the operator to ClickHouse instances. In seconds.
513 # Timout to perform SQL query from the operator to ClickHouse instances. In seconds.
515 ################################################
517 ## Addons specifies additional configuration sections
518 ## Should it be called something like "templates"?
520 ################################################
535 ### users.d is global while description depends on CH version which may vary on per-host basis
536 ### In case of global-ness this may be better to implement via auto-templates
538 ### As a solution, this may be applied on the whole cluster based on any of its hosts
540 ### What to do when host is just created? CH version is not known prior to CH started and user config is required before CH started.
541 ### We do not have any info about the cluster on initial creation
544 "{clickhouseOperatorUser}/access_management": 1
545 "{clickhouseOperatorUser}/named_collection_control": 1
546 "{clickhouseOperatorUser}/show_named_collections": 1
547 "{clickhouseOperatorUser}/show_named_collections_secrets": 1
557 clickhouse_operator/format_display_secrets_in_show_and_select: 1
561 ## this may be added on per-host basis into host's conf.d folder
563 display_secrets_in_show_and_select: 1
565 #################################################
567 ## Metrics collection
569 ################################################
571 # Timeouts used to limit connection and queries from the metrics exporter to ClickHouse instances
572 # Specified in seconds.
574 # Timeout used to limit metrics collection request. In seconds.
575 # Upon reaching this timeout metrics collection is aborted and no more metrics are collected in this cycle.
576 # All collected metrics are returned.
578 # Regexp to match tables in system database to fetch metrics from.
579 # Multiple tables can be matched using regexp. Matched tables are merged using merge() table function.
580 # Default is "^(metrics|custom_metrics)$" which fetches from both system.metrics and system.custom_metrics.
581 tablesRegexp: "^(metrics|custom_metrics)$"
582 # List of regexps to match ClickHouse metrics to exclude from collection/export.
583 # Regexps match internal metric names before Prometheus normalization and prefixing.
584 # Default is the per-CPU OS metrics filter shown below; set to [] to disable.
586 - "^metric\\.(OS.*CPU[0-9]+|CPUFrequencyMHz_[0-9]+)$"
589 ################################################
591 ## Configuration files section
593 ################################################
595 # Each 'path' can be either absolute or relative.
596 # In case path is absolute - it is used as is
597 # In case path is relative - it is relative to the folder where configuration file you are reading right now is located.
599 # Path to the folder where Keeper configuration files common for all instances within a CHK are located.
600 common: chk/keeper_config.d
601 # Path to the folder where Keeper configuration files unique for each instance (host) within a CHK are located.
603 # Path to the folder where Keeper configuration files with users' settings are located.
604 # Files are common for all instances within a CHI.
606 ################################################
610 ## Operator-wide security toggles. All fields default to unset / permissive so
611 ## upgrades from earlier versions preserve identical behavior. Set explicit
612 ## values here to tighten the operator's outbound TLS posture; CHIs may further
613 ## override per-cluster via spec.configuration.clusters[].security.
615 ## Three orthogonal axes govern this section:
616 ## 1. security.policy — TLS-hardening master switch (Permissive | Enforced)
617 ## 2. security.fips.enforced — FIPS cryptographic-module gate (bool)
618 ## 3. security.images.policy — Workload image-tag governance (Permissive | FIPSRequired)
619 ## Each axis is independent: enabling one does not enable the others.
621 ## See docs/chi-examples/70-chop-config.yaml for a fully-annotated example
622 ## and docs/security_hardening.md for the design + per-knob semantics.
624 ################################################
628 # Strict | None | "" (preserve legacy InsecureSkipVerify=true)
630 # "1.2" | "1.3" | "" (Go stdlib default)
632 # SNI / cert-name override; default = dial host
634 # Inline PEM CA bundle (or base64-wrapped)
636 # Alternate source — Secret in operator namespace. Mutually exclusive
637 # with the inline rootCA above. Empty `name` = not used (no-op).
638 # When `key` is empty, the operator tries "ca.crt" then "tls.crt".
648 # Strict refuses an insecure kubeconfig at startup
650 # Floors the K8s API client transport TLS version; coerced to 1.3 under FIPS/Enforced
653 # Plain (default) | Secure (loopback + X-CHOP-Token)
655 # Defaults to 127.0.0.1 when mode=Secure
657 # Defaults to /etc/clickhouse-operator-ipc/token
659 # Operator-wide TLS-hardening master switch. ONLY governs TLS posture across
660 # CH / ZK / K8s transports — NOT the FIPS cryptographic-module gate (see
661 # security.fips.enforced below for that, orthogonal axis).
663 # Permissive (default) preserves 0.27.0 behavior — no coercion, no rejection.
664 # Enforced coerces all TLS knobs above to their Strict positions at startup:
665 # - clickhouse.tls.verify=Strict, clickhouse.tls.minVersion=1.3
666 # - zookeeper.tls.verify=Strict, zookeeper.tls.minVersion=1.3
667 # - kubernetes.tls.verify=Strict, kubernetes.tls.minVersion=1.3
669 # - clickhouse.access.scheme: http is coerced to https
670 # Enforced also rejects CHIs that cannot be served in a hardened posture
671 # (e.g. plaintext external ZooKeeper, ZK digest auth).
673 # Independent of the Go FIPS toolchain — works on non-FIPS builds for pure
674 # TLS hardening. Combine with security.fips.enforced=true for full FIPS
675 # cryptographic-module enforcement.
677 # FIPS cryptographic-module enforcement. Orthogonal to security.policy.
678 # Default operator and metrics-exporter images are FIPS-compatible —
679 # built with GOFIPS140=v1.0.0 and run with GODEBUG=fips140=on, so
680 # crypto/fips140.Enabled() returns true at runtime.
681 # When enforced=true, the operator Fatals at startup unless the binary
682 # reports crypto/fips140 Enabled — guards against accidentally running
683 # a non-FIPS rebuild in a hardened deployment.
687 # Workload image-tag governance gate. Today's only non-default value is
688 # FIPSRequired (admission rejects CRs whose CH/Keeper images lack 'fips'
689 # in tag; post-Ready SELECT version() must contain 'fips' or CR aborts).
690 # Orthogonal to security.policy and security.fips.enforced.
691 # See docs/security_hardening_fips.md → "security.images.policy: FIPSRequired"
692 # for the full policy matrix + detection details + recovery procedure.
694 ################################################
696 ## Template(s) management section
698 ################################################
701 # CHI template updates handling policy
702 # Possible policy values:
703 # - ReadOnStart. Accept CHIT updates on the operator's start only.
704 # - ApplyOnNextReconcile. Accept CHIT updates at all time. Apply new CHITs on next regular reconcile of the CHI
705 policy: ApplyOnNextReconcile
706 # Path to the folder where ClickHouseInstallation templates .yaml manifests are located.
707 # Templates are added to the list of all templates and used when CHI is reconciled.
708 # Templates are applied in sorted alpha-numeric order.
709 path: chi/templates.d
710 # NB there is deliberately no `chk:` section here. OperatorConfigTemplate declares only a CHI
711 # field, so any template.chk.* key unmarshals into nothing and is pruned by the chopconf CRD -
712 # it read as a supported knob while doing nothing. CHK templating is not implemented; adding
713 # the keys back requires the Go field and a reader first.
714 ################################################
718 ################################################
720 # Reconcile runtime settings
722 # Max number of concurrent CHI reconciles in progress
723 reconcileCHIsThreadsNumber: 10
724 # Max number of concurrent CHK reconciles in progress
725 reconcileCHKsThreadsNumber: 1
726 # The operator reconciles shards concurrently in each CHI with the following limitations:
727 # 1. Number of shards being reconciled (and thus having hosts down) in each CHI concurrently
728 # can not be greater than 'reconcileShardsThreadsNumber'.
729 # 2. Percentage of shards being reconciled (and thus having hosts down) in each CHI concurrently
730 # can not be greater than 'reconcileShardsMaxConcurrencyPercent'.
731 # 3. The first shard is always reconciled alone. Concurrency starts from the second shard and onward.
732 # Thus limiting number of shards being reconciled (and thus having hosts down) in each CHI by both number and percentage
734 # Max number of concurrent shard reconciles within one cluster in progress
735 reconcileShardsThreadsNumber: 5
736 # Max percentage of concurrent shard reconciles within one cluster in progress
737 reconcileShardsMaxConcurrencyPercent: 50
738 # Reconcile StatefulSet scenario
740 # Create StatefulSet scenario
742 # What to do in case created StatefulSet is not in 'Ready' after `reconcile.statefulSet.update.timeout` seconds
744 # 1. abort - abort the process, do nothing with the problematic StatefulSet, leave it as it is,
745 # do not try to fix or delete or update it, just abort reconcile cycle.
746 # Do not proceed to the next StatefulSet(s) and wait for an admin to assist.
747 # 2. delete - delete newly created problematic StatefulSet and follow 'abort' path afterwards.
748 # 3. ignore - ignore an error, pretend nothing happened, continue reconcile and move on to the next StatefulSet.
750 # Update StatefulSet scenario
752 # How many seconds to wait for created/updated StatefulSet to be 'Ready'
754 # How many seconds to wait between checks/polls for created/updated StatefulSet status
756 # What to do in case updated StatefulSet is not in 'Ready' after `reconcile.statefulSet.update.timeout` seconds
758 # 1. abort - abort the process, do nothing with the problematic StatefulSet, leave it as it is,
759 # do not try to fix or delete or update it, just abort reconcile cycle.
760 # Do not proceed to the next StatefulSet(s) and wait for an admin to assist.
761 # 2. rollback - delete Pod and rollback StatefulSet to previous Generation.
762 # Pod would be recreated by StatefulSet based on rollback-ed StatefulSet configuration.
763 # Follow 'abort' path afterwards.
764 # 3. ignore - ignore an error, pretend nothing happened, continue reconcile and move on to the next StatefulSet.
766 # Recreate StatefulSet scenario
768 # What to do in case operator is in need to recreate StatefulSet?
770 # 1. abort - abort the process, do nothing with the problematic StatefulSet, leave it as it is,
771 # do not try to fix or delete or update it, just abort reconcile cycle.
772 # Do not proceed to the next StatefulSet(s) and wait for an admin to assist.
773 # 2. recreate - proceed and recreate StatefulSet.
775 # Triggered when PVC data loss or missing volumes are detected
777 # Triggered when StatefulSet update fails or StatefulSet is not ready
778 onUpdateFailure: recreate
779 # Reconcile Host scenario
781 # The operator during reconcile procedure should wait for a ClickHouse host to achieve the following conditions:
783 # Whether the operator during reconcile procedure should wait for a ClickHouse host:
784 # - to be excluded from a ClickHouse cluster
785 # - to complete all running queries
786 # - to be included into a ClickHouse cluster
787 # respectfully before moving forward with host reconcile
791 # The operator during reconcile procedure should wait for replicas to catch-up
792 # replication delay a.k.a replication lag for the following replicas
794 # All replicas (new and known earlier) are explicitly requested to wait for replication to catch-up
796 # New replicas only are requested to wait for replication to catch-up
798 # Replication catch-up is considered to be completed as soon as replication delay
799 # a.k.a replication lag - calculated as "MAX(absolute_delay) FROM system.replicas"
800 # is within this specified delay (in seconds)
803 # Whether the operator during host launch procedure should wait for startup probe to succeed.
804 # In case probe is unspecified wait is assumed to be completed successfully.
805 # Default option value is to do not wait.
807 # Whether the operator during host launch procedure should wait for readiness probe to succeed.
808 # In case probe is unspecified wait is assumed to be completed successfully.
809 # Default option value is to wait.
811 # The operator during reconcile procedure should drop the following entities:
814 # Whether the operator during reconcile procedure should drop replicas when replica is deleted
816 # Whether the operator during reconcile procedure should drop replicas when replica volume is lost
818 # Whether the operator during reconcile procedure should drop active replicas when replica is deleted or recreated
820 ################################################
822 ## Coordination with external systems during reconcile
824 ################################################
827 # How long the operator waits for a referenced ClickHouseKeeper to become ready
828 # before aborting CHI reconcile. In seconds.
830 # Reaction when a referenced CHK resource changes:
831 # none — do nothing (default, backward-compatible)
832 # reconcile — trigger CHI reconcile
833 onKeeperResourceUpdate: none
834 ################################################
836 ## Auto-recovery from aborted reconcile
838 ################################################
840 # Recovery scopes keyed by the CHI .status.status they apply to.
841 # Each scope contains on<Event>: <action> mappings that apply while the CHI
842 # is in that status. Multi-scope design anticipates future states beyond Aborted
843 # (e.g. Failed, Broken).
845 # Recovery for a CHI whose .status.status is Aborted (reconcile did not complete)
846 # when one of its host pods transitions to Ready — auto-resumes the reconcile.
848 # Action when a pod belonging to an Aborted CHI transitions to Ready:
849 # retry (default) — re-enqueue the CHI for reconcile
850 # none — do nothing, CHI stays Aborted
852 # Future events (not yet implemented):
853 # onKeeperReady: retry — retry when a referenced CHK becomes ready
854 # onOperatorRestart: retry — sweep Aborted CHIs on operator startup
855 # Recovery for a CHI whose .status.status is Completed (fully reconciled) when one
856 # of its host pods regresses to Ready=False and stays NotReady (sustained) without
857 # crashing — auto-heals stuck hosts.
859 # Action when a Completed CHI's pod flips Ready=True -> Ready=False and
860 # stays NotReady for at least onPodNotReadyThreshold:
861 # none (default) — do nothing
862 # retry — re-enqueue the CHI so the stuck host is force-restarted
863 # OFF by default: force-recreating a Completed CHI's pod is destructive — it can
864 # interrupt a replica's in-progress recovery and means hard downtime for a
865 # single-replica shard. Opt in with `retry` only where that trade-off is acceptable.
867 # Minimum duration a pod must stay Ready=False before recovery fires, once enabled
868 # (Go duration string; default 5m). Raise it for slow-recovering replicas.
869 onPodNotReadyThreshold: 5m
870 # Future scopes (not yet implemented):
875 # Future global policy knobs (not yet implemented) — flat peers of `onStatus`,
876 # apply across all recovery scopes:
878 # Global kill-switch for auto-recovery:
881 # Cap on consecutive auto-recovery attempts before giving up:
884 # Minimum time between auto-recovery attempts for the same CHI:
887 # Exponential backoff for auto-recovery attempts:
892 ################################################
894 ## Annotations management section
896 ################################################
899 # 1. Propagating annotations from the CHI's `metadata.annotations` to child objects' `metadata.annotations`,
900 # 2. Propagating annotations from the CHI Template's `metadata.annotations` to CHI's `metadata.annotations`,
901 # Include annotations from the following list:
902 # Applied only when not empty. Empty list means "include all, no selection"
904 # Exclude annotations from the following list:
906 ################################################
908 ## Labels management section
910 ################################################
913 # 1. Propagating labels from the CHI's `metadata.labels` to child objects' `metadata.labels`,
914 # 2. Propagating labels from the CHI Template's `metadata.labels` to CHI's `metadata.labels`,
915 # Include labels from the following list:
916 # Applied only when not empty. Empty list means "include all, no selection"
918 # Exclude labels from the following list:
919 # Applied only when not empty. Empty list means "nothing to exclude, no selection"
921 # Whether to append *Scope* labels to StatefulSet and Pod.
922 # Full list of available *scope* labels check in 'labeler.go'
923 # LabelShardScopeIndex
924 # LabelReplicaScopeIndex
926 # LabelCHIScopeCycleSize
927 # LabelCHIScopeCycleIndex
928 # LabelCHIScopeCycleOffset
929 # LabelClusterScopeIndex
930 # LabelClusterScopeCycleSize
931 # LabelClusterScopeCycleIndex
932 # LabelClusterScopeCycleOffset
934 ################################################
936 ## Metrics management section
938 ################################################
942 ################################################
944 ## Status management section
946 ################################################
953 ################################################
955 ## StatefulSet management section
957 ################################################
959 revisionHistoryLimit: 0
960 ################################################
962 ## Pod management section
964 ################################################
966 # Grace period for Pod termination.
967 # How many seconds to wait between sending
968 # SIGTERM and SIGKILL during Pod termination process.
969 # Increase this number is case of slow shutdown.
970 terminationGracePeriod: 30
971 ################################################
973 ## Log parameters section
975 ################################################
978 alsologtostderr: "false"
984 001-templates.json.example: |
986 "apiVersion": "clickhouse.altinity.com/v1",
987 "kind": "ClickHouseInstallationTemplate",
989 "name": "01-default-volumeclaimtemplate"
993 "volumeClaimTemplates": [
995 "name": "chi-default-volume-claim-template",
1010 "name": "chi-default-oneperhost-pod-template",
1011 "distribution": "OnePerHost",
1015 "name": "clickhouse",
1016 "image": "clickhouse/clickhouse-server:23.8",
1020 "containerPort": 8123
1024 "containerPort": 9000
1027 "name": "interserver",
1028 "containerPort": 9009
1039 default-pod-template.yaml.example: |
1040 apiVersion: "clickhouse.altinity.com/v1"
1041 kind: "ClickHouseInstallationTemplate"
1043 name: "default-oneperhost-pod-template"
1047 - name: default-oneperhost-pod-template
1048 distribution: "OnePerHost"
1049 default-storage-template.yaml.example: |
1050 apiVersion: "clickhouse.altinity.com/v1"
1051 kind: "ClickHouseInstallationTemplate"
1053 name: "default-storage-template-2Gi"
1056 volumeClaimTemplates:
1057 - name: default-storage-template-2Gi
1065 Templates in this folder are packaged with an operator and available via 'useTemplate'
1067 01-clickhouse-operator-profile.xml: |
1069 <!-- This file is auto-generated -->
1070 <!-- Do not edit this file - all changes would be lost -->
1071 <!-- Edit appropriate template in the following folder: -->
1072 <!-- deploy/builder/templates-config -->
1076 # Template parameters available:
1080 <!-- clickhouse-operator user is generated by the operator based on config.yaml in runtime -->
1082 <clickhouse_operator>
1083 <log_queries>0</log_queries>
1084 <skip_unavailable_shards>1</skip_unavailable_shards>
1085 <http_connection_timeout>10</http_connection_timeout>
1086 <max_concurrent_queries_for_all_users>0</max_concurrent_queries_for_all_users>
1087 <os_thread_priority>0</os_thread_priority>
1088 </clickhouse_operator>
1091 02-clickhouse-default-profile.xml: |-
1093 <!-- This file is auto-generated -->
1094 <!-- Do not edit this file - all changes would be lost -->
1095 <!-- Edit appropriate template in the following folder: -->
1096 <!-- deploy/builder/templates-config -->
1101 <os_thread_priority>2</os_thread_priority>
1102 <log_queries>1</log_queries>
1103 <connect_timeout_with_failover_ms>1000</connect_timeout_with_failover_ms>
1104 <distributed_aggregation_memory_efficient>1</distributed_aggregation_memory_efficient>
1105 <parallel_view_processing>1</parallel_view_processing>
1106 <do_not_merge_across_partitions_select_final>1</do_not_merge_across_partitions_select_final>
1107 <load_balancing>nearest_hostname</load_balancing>
1108 <prefer_localhost_replica>0</prefer_localhost_replica>
1109 <!-- materialize_ttl_recalculate_only>1</materialize_ttl_recalculate_only> 21.10 and above -->
1113 keeperConfdFiles: null
1115 01-keeper-01-default-config.xml: |
1117 <!-- This file is auto-generated -->
1118 <!-- Do not edit this file - all changes would be lost -->
1119 <!-- Edit appropriate template in the following folder: -->
1120 <!-- deploy/builder/templates-config -->
1123 <asynchronous_metrics_keeper_metrics_only>1</asynchronous_metrics_keeper_metrics_only>
1125 <coordination_settings>
1126 <async_replication>1</async_replication>
1127 <min_session_timeout_ms>10000</min_session_timeout_ms>
1128 <operation_timeout_ms>10000</operation_timeout_ms>
1129 <raft_logs_level>information</raft_logs_level>
1130 <session_timeout_ms>100000</session_timeout_ms>
1131 <use_xid_64>1</use_xid_64>
1132 </coordination_settings>
1133 <hostname_checks_enabled>true</hostname_checks_enabled>
1134 <log_storage_path>/var/lib/clickhouse-keeper/coordination/logs</log_storage_path>
1135 <snapshot_storage_path>/var/lib/clickhouse-keeper/coordination/snapshots</snapshot_storage_path>
1136 <storage_path>/var/lib/clickhouse-keeper</storage_path>
1137 <tcp_port>2181</tcp_port>
1139 Four-letter-word command allowlist.
1141 Set explicitly to the upstream-default list so the operator-rendered
1142 liveness probe (which sends `ruok` over TCP and expects `imok`) keeps
1143 working even if a user adds their own keeper_server settings.
1145 Without this, a user override that restricts the allowlist
1146 (e.g. `four_letter_word_white_list: "mntr,stat"` for security)
1147 would silently disable `ruok` → liveness probe always fails → CrashLoopBackOff.
1149 The list mirrors ClickHouse Keeper's compiled-in default; users who want a
1150 stricter list can override this value, but they must keep `ruok` if they
1151 also use the default operator probes.
1153 <four_letter_word_white_list>conf,cons,crst,envi,ruok,srst,srvr,stat,wchs,dirs,mntr,isro</four_letter_word_white_list>
1155 <listen_host>::</listen_host>
1156 <listen_host>0.0.0.0</listen_host>
1157 <listen_try>1</listen_try>
1159 <console>1</console>
1160 <level>information</level>
1162 <max_connections>4096</max_connections>
1164 01-keeper-02-readiness.xml: |
1166 <!-- This file is auto-generated -->
1167 <!-- Do not edit this file - all changes would be lost -->
1168 <!-- Edit appropriate template in the following folder: -->
1169 <!-- deploy/builder/templates-config -->
1176 <endpoint>/ready</endpoint>
1181 01-keeper-03-enable-reconfig.xml: |-
1183 <!-- This file is auto-generated -->
1184 <!-- Do not edit this file - all changes would be lost -->
1185 <!-- Edit appropriate template in the following folder: -->
1186 <!-- deploy/builder/templates-config -->
1190 <enable_reconfiguration>false</enable_reconfiguration>
1193 keeperTemplatesdFiles:
1195 Templates in this folder are packaged with an operator and available via 'useTemplate'
1196 keeperUsersdFiles: null
1197# additionalResources -- list of additional resources to create (processed via `tpl` function),
1198# useful for create ClickHouse clusters together with clickhouse-operator.
1199# check `kubectl explain chi` for details
1200additionalResources: []
1205# name: {{ include "altinity-clickhouse-operator.fullname" . }}-cm
1206# namespace: {{ include "altinity-clickhouse-operator.namespace" . }}
1211# name: {{ include "altinity-clickhouse-operator.fullname" . }}-s
1212# namespace: {{ include "altinity-clickhouse-operator.namespace" . }}
1216# apiVersion: clickhouse.altinity.com/v1
1217# kind: ClickHouseInstallation
1219# name: {{ include "altinity-clickhouse-operator.fullname" . }}-chi
1220# namespace: {{ include "altinity-clickhouse-operator.namespace" . }}
1229 # dashboards.enabled -- provision grafana dashboards as configMaps (can be synced by grafana dashboards sidecar https://github.com/grafana/helm-charts/blob/grafana-8.3.4/charts/grafana/values.yaml#L778 )
1231 # dashboards.additionalLabels -- labels to add to a secret with dashboards
1233 # dashboards.additionalLabels.grafana_dashboard - will watch when official grafana helm chart sidecar.dashboards.enabled=true
1234 grafana_dashboard: ""
1235 # dashboards.annotations -- annotations to add to a secret with dashboards
1237 # dashboards.annotations.grafana_folder -- folder where will place dashboards, requires define values in official grafana helm chart sidecar.dashboards.folderAnnotation: grafana_folder
1238 grafana_folder: clickhouse-operator