diff --git a/.release-please-manifest.json b/.release-please-manifest.json index e4df6aab..233e87d2 100644 --- a/.release-please-manifest.json +++ b/.release-please-manifest.json @@ -1,3 +1,3 @@ { - ".": "0.0.17" + ".": "0.0.18" } diff --git a/CHANGELOG.md b/CHANGELOG.md index c0f59e16..03b87bfb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,47 @@ # Changelog +## [0.0.18](https://github.com/cloudnative-pg/klio/compare/v0.0.17...v0.0.18) (2026-08-13) + + +### Features + +* **core:** Export Kopia traces via OTLP/gRPC ([EnterpriseDB/klio#1679](https://github.com/EnterpriseDB/klio/issues/1679)) ([245da93](https://github.com/cloudnative-pg/klio/commit/245da93ebdc24d0f301ceaa26134d143ac1cc732)) +* **ip:** Contribute Klio to CloudNativePG under the Apache License 2.0 ([EnterpriseDB/klio#1841](https://github.com/EnterpriseDB/klio/issues/1841)) ([a3dcd13](https://github.com/cloudnative-pg/klio/commit/a3dcd1377934dbbe6f025347488aa8c5918c41bd)) +* **observability:** Fill Grafana dashboard gaps for WAL and backup metrics ([EnterpriseDB/klio#1793](https://github.com/EnterpriseDB/klio/issues/1793)) ([c88d2f8](https://github.com/cloudnative-pg/klio/commit/c88d2f8fe1ee3b59360c83da0900667907f09694)) +* **operator:** Add preflight check operator certification ([EnterpriseDB/klio#1767](https://github.com/EnterpriseDB/klio/issues/1767)) ([295ea20](https://github.com/cloudnative-pg/klio/commit/295ea203c1e3dd9e9f1dc9b942944a47ace91528)) +* Switch from UBI images to Debian based ones ([#90](https://github.com/cloudnative-pg/klio/issues/90)) ([af333a5](https://github.com/cloudnative-pg/klio/commit/af333a5ecde3a469df95560384fe67eb68f2fc5f)) + + +### Bug Fixes + +* **ci:** Upload openshift e2e logs from the correct path ([EnterpriseDB/klio#1808](https://github.com/EnterpriseDB/klio/issues/1808)) ([e2fb58e](https://github.com/cloudnative-pg/klio/commit/e2fb58ef0d4dcc7ed3c67ac8c972d9770366cc90)) +* **core:** Archive partial WAL segments to tier2 and restore from them ([EnterpriseDB/klio#1747](https://github.com/EnterpriseDB/klio/issues/1747)) ([924914e](https://github.com/cloudnative-pg/klio/commit/924914e2df1ac73650ad455da32bd7e1a605a887)) +* **deps:** Align semconv to v1.43.0 in core and operator ([cd74155](https://github.com/cloudnative-pg/klio/commit/cd74155fa65e1eca9a7428b33fe7ac8f67f0c0f0)) +* **deps:** Pin jsm.go to main for StreamPager cross-delivery fix ([EnterpriseDB/klio#1806](https://github.com/EnterpriseDB/klio/issues/1806)) ([ae18c70](https://github.com/cloudnative-pg/klio/commit/ae18c701ce755b5d07e9379e113ef3459af9c80b)) +* **deps:** Update all non-major go dependencies ([eea11e5](https://github.com/cloudnative-pg/klio/commit/eea11e5b1a32ee1852726ab9ad1a08c25f69e613)) +* **deps:** Update all non-major go dependencies ([b501f13](https://github.com/cloudnative-pg/klio/commit/b501f132c371abb63eb22ddd1ac48ea0c74566d8)) +* **deps:** Update all non-major go dependencies ([EnterpriseDB/klio#1780](https://github.com/EnterpriseDB/klio/issues/1780)) ([c76337c](https://github.com/cloudnative-pg/klio/commit/c76337c45833d14b743fa9967bfd9fbf72409bad)) +* **deps:** Update all non-major go dependencies ([EnterpriseDB/klio#1797](https://github.com/EnterpriseDB/klio/issues/1797)) ([21c34a4](https://github.com/cloudnative-pg/klio/commit/21c34a4727e29efc45ea901a0cc0c3ca4f331008)) +* **deps:** Update all non-major go dependencies ([EnterpriseDB/klio#1833](https://github.com/EnterpriseDB/klio/issues/1833)) ([8bbd83f](https://github.com/cloudnative-pg/klio/commit/8bbd83fbf5e1c29983902a0f04a91459458caceb)) +* **deps:** Update all non-major go dependencies ([EnterpriseDB/klio#1845](https://github.com/EnterpriseDB/klio/issues/1845)) ([762f70d](https://github.com/cloudnative-pg/klio/commit/762f70d20d2124db797e117003c31b66f21fcb8d)) +* **deps:** Update all non-major go dependencies ([EnterpriseDB/klio#1848](https://github.com/EnterpriseDB/klio/issues/1848)) ([ef9330e](https://github.com/cloudnative-pg/klio/commit/ef9330e8602c567b3feee4acc4f4551f57ee6e97)) +* **deps:** Update documentation dependencies to v3.10.2 ([EnterpriseDB/klio#1798](https://github.com/EnterpriseDB/klio/issues/1798)) ([14c4161](https://github.com/cloudnative-pg/klio/commit/14c4161aec7d10fe4585e8e5555e3928f0cc4fbf)) +* **deps:** Update github.com/cloudnative-pg/cloudnative-pg/tests digest to 0552b9c ([EnterpriseDB/klio#1859](https://github.com/EnterpriseDB/klio/issues/1859)) ([e01f8d6](https://github.com/cloudnative-pg/klio/commit/e01f8d612dd2462055b05a38671db1304418955d)) +* **deps:** Update github.com/cloudnative-pg/cloudnative-pg/tests digest to 0603a9c ([eeb3f8f](https://github.com/cloudnative-pg/klio/commit/eeb3f8f224df0a20b58390537f0a722145dbd9a5)) +* **deps:** Update github.com/cloudnative-pg/cloudnative-pg/tests digest to 1a79065 ([EnterpriseDB/klio#1803](https://github.com/EnterpriseDB/klio/issues/1803)) ([42c7f9c](https://github.com/cloudnative-pg/klio/commit/42c7f9c8135ed3ff31fa9318347b06906229ed02)) +* **deps:** Update github.com/cloudnative-pg/cloudnative-pg/tests digest to 2af6cc5 ([EnterpriseDB/klio#1842](https://github.com/EnterpriseDB/klio/issues/1842)) ([a8c97ed](https://github.com/cloudnative-pg/klio/commit/a8c97eda63eefe671bda9b74cb837c2e96044415)) +* **deps:** Update github.com/cloudnative-pg/cloudnative-pg/tests digest to 308f3ea ([EnterpriseDB/klio#1794](https://github.com/EnterpriseDB/klio/issues/1794)) ([2d7ba7b](https://github.com/cloudnative-pg/klio/commit/2d7ba7b0a0be24ec7a58d7a82c05610228c498dc)) +* **deps:** Update github.com/cloudnative-pg/cloudnative-pg/tests digest to 83bfc38 ([EnterpriseDB/klio#1865](https://github.com/EnterpriseDB/klio/issues/1865)) ([e418675](https://github.com/cloudnative-pg/klio/commit/e4186751dae758e45e898802da90ad28e5f76648)) +* **deps:** Update github.com/cloudnative-pg/cloudnative-pg/tests digest to 91c9e02 ([EnterpriseDB/klio#1810](https://github.com/EnterpriseDB/klio/issues/1810)) ([a5c8de1](https://github.com/cloudnative-pg/klio/commit/a5c8de1dac06f2daefeac8416aa323e1b2904e9c)) +* **deps:** Update github.com/cloudnative-pg/cloudnative-pg/tests digest to e39ec6b ([EnterpriseDB/klio#1822](https://github.com/EnterpriseDB/klio/issues/1822)) ([9238238](https://github.com/cloudnative-pg/klio/commit/92382385048336e7de2578a1673fc4e964d5d124)) +* **deps:** Update kubernetes monorepo to v0.36.3 ([EnterpriseDB/klio#1854](https://github.com/EnterpriseDB/klio/issues/1854)) ([4189cfe](https://github.com/cloudnative-pg/klio/commit/4189cfeb9925a0ed3619d268c720132d5f96e5a5)) +* **deps:** Update module github.com/aws/aws-sdk-go-v2/service/s3 to v1.105.2 ([EnterpriseDB/klio#1816](https://github.com/EnterpriseDB/klio/issues/1816)) ([d66d8e1](https://github.com/cloudnative-pg/klio/commit/d66d8e1c60cce79d5394b688a93b5d80668d7ef3)) +* **deps:** Update module github.com/nats-io/nats-server/v2 to v2.14.5 ([#92](https://github.com/cloudnative-pg/klio/issues/92)) ([26d6449](https://github.com/cloudnative-pg/klio/commit/26d6449b7edd740fa2e1a8095804c31eed50cf1a)) +* **deps:** Update module go.yaml.in/yaml/v3 to v3.0.5 ([EnterpriseDB/klio#1874](https://github.com/EnterpriseDB/klio/issues/1874)) ([2415112](https://github.com/cloudnative-pg/klio/commit/241511233e75462bbcde935cd69d46701635903e)) +* **deps:** Update module google.golang.org/grpc to v1.82.1 ([EnterpriseDB/klio#1813](https://github.com/EnterpriseDB/klio/issues/1813)) ([7a8df3c](https://github.com/cloudnative-pg/klio/commit/7a8df3cc948d3a439115ba8d82aec9e734b20c3c)) +* **operator:** Drop stale klio-wal from container allowlist ([#86](https://github.com/cloudnative-pg/klio/issues/86)) ([f71c6af](https://github.com/cloudnative-pg/klio/commit/f71c6aff73fe7751bd4e71fef9fb318ed625a695)), closes [#68](https://github.com/cloudnative-pg/klio/issues/68) +* **wal:** Archive timeline history files to tier-2 ([EnterpriseDB/klio#1762](https://github.com/EnterpriseDB/klio/issues/1762)) ([20afc6f](https://github.com/cloudnative-pg/klio/commit/20afc6f09b34b7feb7aa17f9d96c91b0a8586076)) + ## [0.0.17](https://github.com/EnterpriseDB/klio/compare/v0.0.16...v0.0.17) (2026-07-10) diff --git a/documentation/web/docs/user/_helm_chart_values.md b/documentation/web/docs/user/_helm_chart_values.md index 4274cf15..67797466 100644 --- a/documentation/web/docs/user/_helm_chart_values.md +++ b/documentation/web/docs/user/_helm_chart_values.md @@ -10,11 +10,11 @@ | controllerManager.affinity | object | `{}` | Affinity rules for the operator deployment. | | controllerManager.manager.args | list | `["--metrics-bind-address=:8443","--leader-elect","--health-probe-bind-address=:8081","--plugin-server-cert=/pluginServer/tls.crt","--plugin-server-key=/pluginServer/tls.key","--plugin-client-cert=/pluginClient/tls.crt","--plugin-server-address=:9090"]` | List of command line arguments to pass to the controller manager. | | controllerManager.manager.containerSecurityContext | object | `{"allowPrivilegeEscalation":false,"capabilities":{"drop":["ALL"]}}` | The security context for the controller manager container. | -| controllerManager.manager.env | object | `{"SIDECAR_IMAGE":"ghcr.io/cloudnative-pg/klio:v0.0.17"}` | The environment variables to set in the controller manager container. Set OTEL_* variables here to enable OpenTelemetry export. See the OpenTelemetry Observability page in the documentation. | +| controllerManager.manager.env | object | `{"SIDECAR_IMAGE":"ghcr.io/cloudnative-pg/klio:v0.0.18"}` | The environment variables to set in the controller manager container. Set OTEL_* variables here to enable OpenTelemetry export. See the OpenTelemetry Observability page in the documentation. | | controllerManager.manager.image.pullPolicy | string | `"Always"` | The controller manager container imagePullPolicy. | | controllerManager.manager.image.pullSecrets | list | `[]` | The list of imagePullSecrets. | | controllerManager.manager.image.repository | string | `"ghcr.io/cloudnative-pg/klio-operator"` | The image to use for the controller manager container. | -| controllerManager.manager.image.tag | string | `"v0.0.17"` | The tag to use for the controller manager container image. | +| controllerManager.manager.image.tag | string | `"v0.0.18"` | The tag to use for the controller manager container image. | | controllerManager.manager.livenessProbe | object | `{"httpGet":{"path":"/healthz","port":8081},"initialDelaySeconds":15,"periodSeconds":20}` | Liveness probe configuration. | | controllerManager.manager.readinessProbe | object | `{"httpGet":{"path":"/readyz","port":8081},"initialDelaySeconds":5,"periodSeconds":10}` | Readiness probe configuration. | | controllerManager.manager.resources | object | `{"limits":{"cpu":"500m","memory":"128Mi"},"requests":{"cpu":"10m","memory":"64Mi"}}` | The resources to allocate. | diff --git a/documentation/web/docs/user/helm_chart.mdx b/documentation/web/docs/user/helm_chart.mdx index 9131b988..a6e758c6 100644 --- a/documentation/web/docs/user/helm_chart.mdx +++ b/documentation/web/docs/user/helm_chart.mdx @@ -64,7 +64,7 @@ Deploy the Klio Operator to your cluster: ```sh helm install klio-operator \ oci://ghcr.io/cloudnative-pg/klio-operator-chart \ - --version 0.0.17 \ + --version 0.0.18 \ --namespace cnpg-system \ -f values.yaml ``` @@ -104,7 +104,7 @@ default values, and understand what resources it will create: ```sh helm pull oci://ghcr.io/cloudnative-pg/klio-operator-chart \ - --version 0.0.17 \ + --version 0.0.18 \ --untar ``` @@ -158,7 +158,7 @@ manifests: ```sh helm pull oci://ghcr.io/cloudnative-pg/klio-operator-chart \ - --version 0.0.17 \ + --version 0.0.18 \ --untar ``` @@ -197,7 +197,7 @@ Helm upgrade command: ```sh helm upgrade klio-operator \ oci://ghcr.io/cloudnative-pg/klio-operator-chart \ - --version 0.0.17 \ + --version 0.0.18 \ --namespace cnpg-system \ -f values.yaml ``` @@ -228,7 +228,7 @@ released. Update the `spec.image` field with the new image reference: ```sh kubectl patch server -n \ --type merge \ - -p '{"spec":{"image":"ghcr.io/cloudnative-pg/klio:v0.0.17"}}' + -p '{"spec":{"image":"ghcr.io/cloudnative-pg/klio:v0.0.18"}}' ``` diff --git a/documentation/web/docs/user/klio_server.md b/documentation/web/docs/user/klio_server.md index 92a26bbe..3c8830bd 100644 --- a/documentation/web/docs/user/klio_server.md +++ b/documentation/web/docs/user/klio_server.md @@ -318,7 +318,7 @@ metadata: namespace: default spec: # Container image for the Klio server - image: ghcr.io/cloudnative-pg/klio:v0.0.17 + image: ghcr.io/cloudnative-pg/klio:v0.0.18 imagePullPolicy: IfNotPresent imagePullSecrets: [] # Add image pull secrets if needed @@ -497,7 +497,7 @@ spec: mode: read-only # Container image for the Klio server - image: ghcr.io/cloudnative-pg/klio:v0.0.17 + image: ghcr.io/cloudnative-pg/klio:v0.0.18 imagePullPolicy: IfNotPresent # TLS configuration diff --git a/documentation/web/docs/user/walplayer.md b/documentation/web/docs/user/walplayer.md index 8a75eab3..2e888e6e 100644 --- a/documentation/web/docs/user/walplayer.md +++ b/documentation/web/docs/user/walplayer.md @@ -215,7 +215,7 @@ spec: initContainers: # Generate synthetic WAL files - name: generate-wals - image: ghcr.io/cloudnative-pg/klio:v0.0.17 + image: ghcr.io/cloudnative-pg/klio:v0.0.18 imagePullPolicy: Always command: - /usr/bin/klio @@ -230,7 +230,7 @@ spec: containers: # Play WAL files to the Klio server - name: play-wals - image: ghcr.io/cloudnative-pg/klio:v0.0.17 + image: ghcr.io/cloudnative-pg/klio:v0.0.18 imagePullPolicy: Always command: - /usr/bin/klio diff --git a/documentation/web/versioned_docs/version-0.0.18/developer/_protocol.md b/documentation/web/versioned_docs/version-0.0.18/developer/_protocol.md new file mode 100644 index 00000000..4a1665e2 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/developer/_protocol.md @@ -0,0 +1,572 @@ +# Protocol Documentation + + +## Table of Contents + +- [klio_admin.proto](#klio_admin-proto) + - [DeleteBackupRequest](#klio-wal-v1-DeleteBackupRequest) + - [DeleteBackupResponse](#klio-wal-v1-DeleteBackupResponse) + - [FailedBackup](#klio-wal-v1-FailedBackup) + - [FailedWAL](#klio-wal-v1-FailedWAL) + - [ListBackupsRequest](#klio-wal-v1-ListBackupsRequest) + - [ListBackupsResult](#klio-wal-v1-ListBackupsResult) + - [QueueListFailedBackupsRequest](#klio-wal-v1-QueueListFailedBackupsRequest) + - [QueueListFailedBackupsResponse](#klio-wal-v1-QueueListFailedBackupsResponse) + - [QueueListFailedWALsRequest](#klio-wal-v1-QueueListFailedWALsRequest) + - [QueueListFailedWALsResponse](#klio-wal-v1-QueueListFailedWALsResponse) + - [QueueStatusRequest](#klio-wal-v1-QueueStatusRequest) + - [QueueStatusResponse](#klio-wal-v1-QueueStatusResponse) + - [RefreshRequest](#klio-wal-v1-RefreshRequest) + - [RefreshResult](#klio-wal-v1-RefreshResult) + + - [Tier](#klio-wal-v1-Tier) + + - [Admin](#klio-wal-v1-Admin) + +- [klio_wal.proto](#klio_wal-proto) + - [CloseBackupRequest](#klio-wal-v1-CloseBackupRequest) + - [CloseBackupResult](#klio-wal-v1-CloseBackupResult) + - [ClusterMetadata](#klio-wal-v1-ClusterMetadata) + - [GetMetadataRequest](#klio-wal-v1-GetMetadataRequest) + - [GetRequest](#klio-wal-v1-GetRequest) + - [GetResult](#klio-wal-v1-GetResult) + - [PutRequest](#klio-wal-v1-PutRequest) + - [PutResult](#klio-wal-v1-PutResult) + - [RequestWALStartRequest](#klio-wal-v1-RequestWALStartRequest) + - [RequestWALStartResult](#klio-wal-v1-RequestWALStartResult) + - [ResetWALStreamRequest](#klio-wal-v1-ResetWALStreamRequest) + - [ResetWALStreamResult](#klio-wal-v1-ResetWALStreamResult) + - [StartWALFile](#klio-wal-v1-StartWALFile) + - [WALGap](#klio-wal-v1-WALGap) + + - [WAL](#klio-wal-v1-WAL) + +- [Scalar Value Types](#scalar-value-types) + + + + +

Top

+ +## klio_admin.proto + + + + + +### DeleteBackupRequest +DeleteBackupRequest is the request to delete a backup from the server. + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| backup_name | [string](#string) | | The name of the backup to delete. | +| tiers | [Tier](#klio-wal-v1-Tier) | repeated | The storage tiers from which to delete the backup. At least one tier must be specified. | +| cluster_name | [string](#string) | | The name of the cluster that owns the backup. | + + + + + + + + +### DeleteBackupResponse +DeleteBackupResponse is the response to a backup deletion request. + + + + + + + + +### FailedBackup + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| cluster_name | [string](#string) | | | +| last_attempt_time | [google.protobuf.Timestamp](#google-protobuf-Timestamp) | | | + + + + + + + + +### FailedWAL + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| cluster_name | [string](#string) | | | +| wal_name | [string](#string) | | | +| sequence | [uint64](#uint64) | | | +| last_attempt_time | [google.protobuf.Timestamp](#google-protobuf-Timestamp) | | | + + + + + + + + +### ListBackupsRequest + + + + + + + + + +### ListBackupsResult + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| backup_manifests | [bytes](#bytes) | | JSON-serialized array of backup manifests. Each manifest contains fields like: - id: string - cluster_name: string - timestamp: RFC3339 string - size_bytes: number See klioclient.BackupManifest for the canonical structure. We use JSON bytes here to avoid duplicating the internal type definition and conversion logic, as this is a local admin API. | + + + + + + + + +### QueueListFailedBackupsRequest + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| cluster_name | [string](#string) | optional | | + + + + + + + + +### QueueListFailedBackupsResponse + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| backups | [FailedBackup](#klio-wal-v1-FailedBackup) | repeated | | + + + + + + + + +### QueueListFailedWALsRequest + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| cluster_name | [string](#string) | optional | | + + + + + + + + +### QueueListFailedWALsResponse + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| wals | [FailedWAL](#klio-wal-v1-FailedWAL) | repeated | | + + + + + + + + +### QueueStatusRequest + + + + + + + + + +### QueueStatusResponse + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| pending_backups | [uint64](#uint64) | | Number of backup synchronization tasks pending in the queue | +| pending_wals | [uint64](#uint64) | | Number of WAL relay tasks pending in the queue | + + + + + + + + +### RefreshRequest + + + + + + + + + +### RefreshResult + + + + + + + + + + + +### Tier +Tier represents a storage tier in the backup system. + +| Name | Number | Description | +| ---- | ------ | ----------- | +| TIER_UNSPECIFIED | 0 | TIER_UNSPECIFIED is the default value and should not be used. | +| TIER_1 | 1 | TIER_1 represents the local cache storage. | +| TIER_2 | 2 | TIER_2 represents the object storage. | + + + + + + + + + +### Admin + + +| Method Name | Request Type | Response Type | Description | +| ----------- | ------------ | ------------- | ------------| +| Refresh | [RefreshRequest](#klio-wal-v1-RefreshRequest) | [RefreshResult](#klio-wal-v1-RefreshResult) | Invoked to refresh the policies and the cache of the Kopia server | +| ListBackups | [ListBackupsRequest](#klio-wal-v1-ListBackupsRequest) | [ListBackupsResult](#klio-wal-v1-ListBackupsResult) | List every backup on the server | +| QueueListFailedBackups | [QueueListFailedBackupsRequest](#klio-wal-v1-QueueListFailedBackupsRequest) | [QueueListFailedBackupsResponse](#klio-wal-v1-QueueListFailedBackupsResponse) | List backups failed to be processed from the queue | +| QueueListFailedWALs | [QueueListFailedWALsRequest](#klio-wal-v1-QueueListFailedWALsRequest) | [QueueListFailedWALsResponse](#klio-wal-v1-QueueListFailedWALsResponse) | List WAL files failed to be processed from the queue | +| QueueStatus | [QueueStatusRequest](#klio-wal-v1-QueueStatusRequest) | [QueueStatusResponse](#klio-wal-v1-QueueStatusResponse) | Get the status of the task queue (pending backups and WALs) | +| DeleteBackup | [DeleteBackupRequest](#klio-wal-v1-DeleteBackupRequest) | [DeleteBackupResponse](#klio-wal-v1-DeleteBackupResponse) | Delete a backup from the server | + + + + + + +

Top

+ +## klio_wal.proto + + + + + +### CloseBackupRequest +This is sent to the WAL server every time a backup has +been completed. + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| cluster_name | [string](#string) | | The name of the cluster. | +| backup_name | [string](#string) | | The name of the backup. | +| timeline | [int32](#int32) | | The backup timeline | +| start_wal | [string](#string) | | the first WAL required to restore this backup. | +| end_wal | [string](#string) | | The last WAL required to restore this backup. | +| segment_size | [uint64](#uint64) | | The size of a WAL segment. Needed to generate the sequence of WAL files between the start and the end. | +| send_to_tier2 | [bool](#bool) | | Require this backup to be sent to tier2. | +| tier2_retention_policy | [string](#string) | | When present, set the tier2 retention policy to the specified JSON-serialized policy. | + + + + + + + + +### CloseBackupResult +This is sent by the WAL server in response to a CloseBackupRequest +message. + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| tier2_schedule | [bool](#bool) | | True when the backup has been scheduled to be synchronized to tier2 storage | +| missing_wal_files | [string](#string) | repeated | List of WAL files needed by this backup but still not uploaded to tier1 | + + + + + + + + +### ClusterMetadata +The following messages are written in the cluster metadata +file + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| system_id | [string](#string) | | The system ID of the current cluster | +| gaps | [WALGap](#klio-wal-v1-WALGap) | repeated | The gaps we are aware of in the collected WALs. | + + + + + + + + +### GetMetadataRequest + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| cluster_name | [string](#string) | | | + + + + + + + + +### GetRequest + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| cluster_name | [string](#string) | | | +| wal_name | [string](#string) | | | + + + + + + + + +### GetResult + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| wal_block | [bytes](#bytes) | | | +| segment_size | [uint64](#uint64) | | | + + + + + + + + +### PutRequest + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| cluster_name | [string](#string) | | | +| wal_name | [string](#string) | | | +| wal_block | [bytes](#bytes) | | | +| segment_size | [uint64](#uint64) | | | +| send_to_tier2 | [bool](#bool) | | | + + + + + + + + +### PutResult + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| written_size | [uint64](#uint64) | | | + + + + + + + + +### RequestWALStartRequest + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| cluster_name | [string](#string) | | This is the cluster name | +| system_id | [string](#string) | | This is the system ID | +| current_wal_name | [string](#string) | | This is the current WAL name that is being written by PostgreSQL. If empty, the start WAL name will be found by looking at the stored WAL files. | + + + + + + + + +### RequestWALStartResult + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| wal_name | [string](#string) | | The WAL file where the client is expected to start streaming. | + + + + + + + + +### ResetWALStreamRequest + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| cluster_name | [string](#string) | | This is the cluster name | +| system_id | [string](#string) | | This is the system ID | +| current_wal_name | [string](#string) | | This is the current WAL name that is being written by PostgreSQL. If empty, the start WAL name will be found by looking at the stored WAL files. | + + + + + + + + +### ResetWALStreamResult + + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| wal_name | [string](#string) | | The WAL file where the client is expected to start streaming. | + + + + + + + + +### StartWALFile +The following messages are used to write a WAL file +in the Klio WAL Storage area + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| klio_version | [uint64](#uint64) | | | +| file_length | [uint64](#uint64) | | | + + + + + + + + +### WALGap +WALGap is a know gap in the WAL collection process. +This is usually caused by an invocation of the reset-lsn Klio +feature. + + +| Field | Type | Label | Description | +| ----- | ---- | ----- | ----------- | +| ts | [google.protobuf.Timestamp](#google-protobuf-Timestamp) | | When this gap was detected and created. | +| start | [string](#string) | | When the gap started. | +| end | [string](#string) | | When the gap ends. | + + + + + + + + + + + + + + +### WAL + + +| Method Name | Request Type | Response Type | Description | +| ----------- | ------------ | ------------- | ------------| +| Put | [PutRequest](#klio-wal-v1-PutRequest) stream | [PutResult](#klio-wal-v1-PutResult) | | +| Get | [GetRequest](#klio-wal-v1-GetRequest) | [GetResult](#klio-wal-v1-GetResult) stream | | +| GetMetadata | [GetMetadataRequest](#klio-wal-v1-GetMetadataRequest) | [ClusterMetadata](#klio-wal-v1-ClusterMetadata) | | +| RequestWALStart | [RequestWALStartRequest](#klio-wal-v1-RequestWALStartRequest) | [RequestWALStartResult](#klio-wal-v1-RequestWALStartResult) | | +| ResetWALStream | [ResetWALStreamRequest](#klio-wal-v1-ResetWALStreamRequest) | [ResetWALStreamResult](#klio-wal-v1-ResetWALStreamResult) | | +| CloseBackup | [CloseBackupRequest](#klio-wal-v1-CloseBackupRequest) | [CloseBackupResult](#klio-wal-v1-CloseBackupResult) | | + + + + + +## Scalar Value Types + +| .proto Type | Notes | C++ | Java | Python | Go | C# | PHP | Ruby | +| ----------- | ----- | --- | ---- | ------ | -- | -- | --- | ---- | +| double | | double | double | float | float64 | double | float | Float | +| float | | float | float | float | float32 | float | float | Float | +| int32 | Uses variable-length encoding. Inefficient for encoding negative numbers – if your field is likely to have negative values, use sint32 instead. | int32 | int | int | int32 | int | integer | Bignum or Fixnum (as required) | +| int64 | Uses variable-length encoding. Inefficient for encoding negative numbers – if your field is likely to have negative values, use sint64 instead. | int64 | long | int/long | int64 | long | integer/string | Bignum | +| uint32 | Uses variable-length encoding. | uint32 | int | int/long | uint32 | uint | integer | Bignum or Fixnum (as required) | +| uint64 | Uses variable-length encoding. | uint64 | long | int/long | uint64 | ulong | integer/string | Bignum or Fixnum (as required) | +| sint32 | Uses variable-length encoding. Signed int value. These more efficiently encode negative numbers than regular int32s. | int32 | int | int | int32 | int | integer | Bignum or Fixnum (as required) | +| sint64 | Uses variable-length encoding. Signed int value. These more efficiently encode negative numbers than regular int64s. | int64 | long | int/long | int64 | long | integer/string | Bignum | +| fixed32 | Always four bytes. More efficient than uint32 if values are often greater than 2^28. | uint32 | int | int | uint32 | uint | integer | Bignum or Fixnum (as required) | +| fixed64 | Always eight bytes. More efficient than uint64 if values are often greater than 2^56. | uint64 | long | int/long | uint64 | ulong | integer/string | Bignum | +| sfixed32 | Always four bytes. | int32 | int | int | int32 | int | integer | Bignum or Fixnum (as required) | +| sfixed64 | Always eight bytes. | int64 | long | int/long | int64 | long | integer/string | Bignum | +| bool | | bool | boolean | boolean | bool | bool | boolean | TrueClass/FalseClass | +| string | A string must always contain UTF-8 encoded or 7-bit ASCII text. | string | String | str/unicode | string | string | string | String (UTF-8) | +| bytes | May contain any arbitrary sequence of bytes. | string | ByteString | str | []byte | ByteString | string | String (ASCII-8BIT) | + diff --git a/documentation/web/versioned_docs/version-0.0.18/developer/developers.mdx b/documentation/web/versioned_docs/version-0.0.18/developer/developers.mdx new file mode 100644 index 00000000..b086b6d4 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/developer/developers.mdx @@ -0,0 +1,168 @@ +--- +sidebar_position: 1 +--- + +# Developers guide + +## Development environment prerequisites + +* A [dagger](https://dagger.io/) installation + +## How to get an ephemeral development environment + +For a quick and convenient development experience, you can create an ephemeral +development environment using the `integration:devenv` task from the Taskfile: + +```sh +task integration:devenv +``` + +This command creates and connects you to an ephemeral development environment +with all the necessary components pre-configured, making it easier to test and +develop Klio features without affecting your local setup. + +Here are the instructions to deploy the Klio server and the CNPG cluster with +the Klio plugin, using manifests under `operator/config/samples`. +The following commands are executed inside the dagger container created from +the previous step: + +```sh +# Deploy the Klio server with the related resources +dagger /apps $ kubectl apply -f ../operator/config/samples/klio_v1alpha1_server.yaml + +# Deploy the CNPG Cluster with a patch for the plugin and the related PluginConfiguration CR +dagger /apps $ kubectl apply -f ../operator/config/samples/cluster-example-plugin.yaml +``` + +## How to test Klio in a kind cluster during development + +During the development phase, it is recommended to test any changes in a test +environment as soon as they are implemented. To perform such a test, a developer +can build and deploy the binaries and the images in a reusable Kubernetes environment. + +This section will guide you through the steps required to deploy the Klio operator and +its resources inside a kind cluster. + +1. Create a kind cluster using the `hack/setup-cluster.sh` script from the CNPG repository. +2. To build and deploy Klio images and operator, exec the command: `KIND_CLUSTER_NAME=$(kind get clusters) task integration:deploy-to-kind` +3. Once the operator is in `Ready` state, you can deploy the resources: + * Klio server: `kubectl apply -f operator/config/samples/klio_v1alpha1_server.yaml` + * CNPG Cluster with Klio plugin and related PluginConfiguration CR: `kubectl apply -f operator/config/samples/cluster-example-plugin.yaml` + The PluginConfiguration CR name should be equal to the value of `ref` parameter, + set inside the Cluster manifest `.spec.plugins` section. + + ```yaml + plugins: + - name: klio.cnpg.io + enabled: true + parameters: + pluginConfigurationRef: client-config-example-plugin + ``` + +Repeat step 2 to deploy the new changes applied to the code, and reset the operator. +Then rollout the new binary in both server and cluster pods: + +1. delete the Klio Server pod: `kubectl delete pod server-sample-klio-0` +2. delete the Cluster's pods: `kubectl delete pod -l cnpg.io/cluster=cluster-example` +3. delete the PluginConfiguration CR: `kubectl delete pluginconfiguration client-config-example-plugin` + +### How to get performance profiles with pprof + +1. Open a port forward connection to port 6061: `kubectl port-forward pod/server-sample-klio-0 6061` +2. run pprof tool, e.g.: `go tool pprof -http=:9999 http://localhost:6061/debug/pprof/profile\?seconds=30` + +## How to recreate the GRPC stub and skeleton + +```sh +dagger call protoc --source . -o internal/grpc +``` + +## How to setup manually a Klio server + +Given the following configuration file in `/home/klio/.klio.yaml`: + +```yaml +tier1: + encryption_key: thiskey + base: + repository: /home/klio/klio_data/base + cache: /home/klio/klio_data/base + listen_address: 0.0.0.0:52000 + wal: + listen_address: 0.0.0.0:52001 + path: /home/klio/klio_data/wals +tls: + client_ca_cert: /home/klio/ca-cert.pem + cert: /home/klio/server.crt + key: /home/klio/server.key +``` + +A new Klio repository for both WAL and base backups can be bootstrapped with: + +```sh +openssl genrsa -out ca-key.pem 2048 + +openssl req -new -key ca-key.pem -out ca-csr.pem -subj "/CN=KlioClientsCA" + +openssl x509 -req -in ca-csr.pem -signkey ca-key.pem -out ca-cert.pem -days 365 -extfile <(printf "basicConstraints=CA:TRUE,pathlen:0\nkeyUsage=critical,cRLSign,keyCertSign") + +openssl req -x509 -newkey rsa:4096 -sha256 -days 3650 \ + -nodes -keyout server.key -out server.crt -subj "/CN=klio-server" \ + -addext "subjectAltName=DNS:klio-server,IP:127.0.0.1" + +~/klio server initialize +``` + +The Klio server hosts both the WAL streaming server and the base backup +server in a single process. Start them with: + +```sh +~/klio server start +``` + +## How to setup manually a Klio client + +Given the following configuration file in `/var/lib/postgresql/.klio.yaml`: + +```yaml +client: + cluster_name: pg-dev-1 + wal: + server_cert_path: /home/klio/server.crt + client_cert_path: /home/klio/client-cert.pem + client_key_path: /home/klio/client-key.pem + address: klio-server:52000 + + base: + url: https://klio-server:52001 + server_cert_path: /home/klio/server.crt + client_cert_path: /home/klio/client-cert.pem + client_key_path: /home/klio/client-key.pem + +source: + dsn: "user=postgres replication=yes" + standard_dsn: "dbname=postgres" + slot: klio +``` + +The client certificates can be created with: + +```sh +openssl genrsa -out client-key.pem 2048 + +openssl req -new -key client-key.pem -out client-csr.pem -subj "/CN=klio@cluster-example" + +cat < client-ext.cnf +[ req ] +distinguished_name = req_distinguished_name + +[ req_distinguished_name ] +CN = klio@cluster-example + +[ usr_cert ] +keyUsage = critical, digitalSignature +extendedKeyUsage = clientAuth +EOF + +openssl x509 -req -in client-csr.pem -CA ca-cert.pem -CAkey ca-key.pem -CAcreateserial -out client-cert.pem -days 365 -extfile client-ext.cnf -extensions usr_cert +``` diff --git a/documentation/web/versioned_docs/version-0.0.18/developer/kopia_internals.md b/documentation/web/versioned_docs/version-0.0.18/developer/kopia_internals.md new file mode 100644 index 00000000..462bdf20 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/developer/kopia_internals.md @@ -0,0 +1,460 @@ +# Kopia Internals + +This document explains Kopia's internal architecture, storage systems, and core mechanisms like sessions and checkpoints. + +## Table of Contents + +1. [Storage Architecture](#storage-architecture) +2. [Sessions: The Write Protection Mechanism](#sessions-the-write-protection-mechanism) +3. [Checkpoints: Periodic Data Protection](#checkpoints-periodic-data-protection) + +--- + +## Storage Architecture + +Kopia uses two complementary storage systems: + +1. **Content-Addressable Storage**: For backup data (files, directories) +2. **Label-Addressable Storage**: For metadata (snapshots, policies) + +### Content-Addressable Storage + +Backup data is stored using content-addressable storage: + +1. **Contents**: Fixed-size chunks of data, identified by their hash (content ID) +2. **Pack Blobs**: Multiple contents are grouped into pack blobs for efficient storage +3. **Index Blobs**: Map content IDs to their location within pack blobs +4. **Objects**: Higher-level structures (files, directories) composed of multiple contents + +### Pack Blob Types: Segregation by Content Type + +Pack blobs are segregated based on the type of content they contain. The content ID prefix determines which pack type is used: + +```go +// From content_manager.go +func packPrefixForContentID(contentID ID) blob.ID { + if contentID.HasPrefix() { + return PackBlobIDPrefixSpecial // "q" packs + } + return PackBlobIDPrefixRegular // "p" packs +} +``` + +| Pack Prefix | Blob Name Pattern | Content Type | Content ID Examples | +|-------------|-------------------|--------------|---------------------| +| `p` (regular) | `p7f8a9b0-session1` | File data chunks | `abc123def...` (no prefix) | +| `q` (special) | `q3c4d5e6-session2` | Metadata | `mabc123...` (manifests), `kabc123...` (other) | + +**Content ID prefixes**: +- No prefix: Raw file data (stored in `p` packs) +- `m`: Manifest contents (stored in `q` packs) +- `k`: Other metadata like policies (stored in `q` packs) +- `n`: Index blob contents (stored separately, not in pack blobs) + +**Why segregation matters**: +- Different caching strategies: Data (`p`) and metadata (`q`) can be cached separately +- Different access patterns: Metadata is accessed frequently, data less so +- GC behavior: System contents in `q` packs (manifests, indexes) are never garbage collected by Snapshot GC + +### Index Blobs: The Content Location Map + +Index blobs are critical for understanding how Kopia finds data and determines what's "referenced": + +**What an index entry contains**: + +```go +// From repo/content/index/info.go +type Info struct { + ContentID ID // Hash-based identifier (e.g., "abc123def...") + PackBlobID blob.ID // Which pack blob contains this content + PackOffset uint32 // Byte offset within the pack blob + PackedLength uint32 // Size in bytes (after compression/encryption) + OriginalLength uint32 // Original size before compression + Deleted bool // Soft-delete flag for GC + TimestampSeconds int64 // When this content was written + CompressionHeaderID // Compression algorithm used + // ... +} +``` + +**How index blobs are loaded and searched**: + +Kopia doesn't know which index blob contains a specific content ID. Instead, it loads **ALL index blobs** into memory and merges them into a single searchable structure: + +```go +// From repo/content/index/merged.go +type Merged []Index // Slice of all loaded index blobs + +// GetInfo searches ALL indexes for a content ID +func (m Merged) GetInfo(id ID, result *Info) (bool, error) { + for _, ndx := range m { + ok, err := ndx.GetInfo(id, &tmp) + if ok { + // Found! Keep the one with highest timestamp + // (handles updates/deletions across multiple indexes) + } + } +} +``` + +**How content lookup works**: + +``` +┌─────────────────────────────────────────────────────────────┐ +│ Content Lookup Process │ +├─────────────────────────────────────────────────────────────┤ +│ │ +│ 1. On repository open: Load ALL index blobs │ +│ Index blobs: n0a1b2c3, n4d5e6f7, n8g9h0i1, ... │ +│ Merge into single searchable structure │ +│ │ +│ 2. Application requests content by ContentID │ +│ ContentID = "abc123def456..." │ +│ │ +│ 3. Search merged index for ContentID │ +│ (searches across all loaded index blobs) │ +│ Found: PackBlobID="p7f8a9b0-session1" │ +│ PackOffset=1024 │ +│ PackedLength=4096 │ +│ │ +│ 4. Read from pack blob at specified location │ +│ Read 4096 bytes from "p7f8a9b0-session1" at offset 1024 │ +│ │ +│ 5. Decrypt and decompress to get original content │ +│ │ +└─────────────────────────────────────────────────────────────┘ +``` + +**Why are there multiple index blobs?** + +Each `Flush()` operation writes a new index blob containing entries for the contents written since the last flush. Over time, this creates many small index blobs. Index compaction (part of maintenance) merges these into larger blobs for efficiency. + +**Index blobs determine which packs are "referenced"**: + +A pack blob is considered **referenced** if ANY index entry points to it via `PackBlobID`. This is how Pack GC decides what to keep: + +```go +// Simplified from content_manager_iterate.go +func IterateUnreferencedPacks() { + // Step 1: Build set of all PackBlobIDs from all index entries + usedPacks := NewSet() + for each indexEntry in allIndexes: + usedPacks.Add(indexEntry.PackBlobID) + + // Step 2: Any pack blob NOT in usedPacks is "unreferenced" + for each packBlob in storage: + if not usedPacks.Contains(packBlob.ID): + // This pack can potentially be deleted + callback(packBlob) +} +``` + +**Key insight**: A pack blob becomes "referenced" when an index blob containing its ID is written. Before the index is written, the pack exists in storage but nothing points to it - it's "unreferenced" and vulnerable to Pack GC. This is precisely why sessions exist. + +### Label-Addressable Storage (Manifests) + +Metadata is stored in **manifests** - JSON documents identified by labels rather than content hash: + +1. **Snapshot Manifests**: Record what was backed up, when, and the root object ID +2. **Policy Manifests**: Define retention rules, compression settings, etc. +3. **Other Manifests**: Maintenance schedules, ACLs, etc. + +Each manifest has: +- A unique ID +- A set of labels (key-value pairs) including a required `type` label +- A modification timestamp +- The actual JSON payload + +``` +┌─────────────────────────────────────────────────────────────┐ +│ Storage Hierarchy │ +├─────────────────────────────────────────────────────────────┤ +│ │ +│ LABEL-ADDRESSABLE (Manifests) │ +│ ───────────────────────────── │ +│ Snapshot Manifest │ +│ ├─ Labels: {type: "snapshot", hostname: "...", ...} │ +│ ├─ RootObjectID: "Iabcdef123..." │ +│ └─ StartTime, EndTime, Stats, etc. │ +│ │ +│ │ │ +│ ▼ │ +│ │ +│ CONTENT-ADDRESSABLE (Data) │ +│ ───────────────────────── │ +│ Root Object (directory manifest) │ +│ └─ References child objects by Object ID │ +│ │ │ +│ ▼ │ +│ Child Objects (files, subdirectories) │ +│ └─ Files are composed of Content IDs │ +│ │ │ +│ ▼ │ +│ Contents (stored in Pack Blobs) │ +│ └─ Located via Index Blobs │ +│ │ +└─────────────────────────────────────────────────────────────┘ +``` + +### GC Roots: What Keeps Data Alive + +Kopia has **two levels of garbage collection**, each with different "roots": + +**Level 1 - Snapshot GC (Content-Level)**: +- **GC Roots**: Snapshot manifests +- **What's protected**: Contents reachable from any snapshot +- **What gets marked deleted**: Contents not reachable from any snapshot + +**Level 2 - Pack GC (Blob-Level)**: +- **GC Roots**: Index blobs (via PackBlobID references) +- **What's protected**: Pack blobs referenced by any index entry +- **What gets deleted**: Pack blobs with no index entries pointing to them + +**Snapshot GC process** (marks contents as deleted): + +1. **Enumerate all snapshot manifests** (from label-addressable storage) +2. **Walk the object tree** starting from each snapshot's root object ID +3. **Mark all reachable content IDs** as "in use" +4. **Any content not marked** is eligible for deletion + +```go +// From snapshotgc/gc.go - simplified +func findInUseContentIDs(ctx context.Context, rep repo.Repository, used *Set) error { + // Step 1: Get all snapshot manifests (GC roots) + manifests, _ := snapshot.LoadSnapshots(ctx, rep, ids) + + // Step 2: Walk each snapshot's object tree + for _, m := range manifests { + root := snapshotfs.SnapshotRoot(rep, m) + // Walk tree, marking all content IDs as used + walker.Process(ctx, root, "") + } +} +``` + +**Important implications**: +- Deleting a snapshot manifest makes its unique data eligible for GC +- Data shared between snapshots remains alive as long as ANY snapshot references it +- Manifests themselves are stored as contents with a special prefix (`m`) +- System contents (manifests, indexes) are never garbage collected + +### The Two-Phase Write Problem + +When writing data, there's a critical window between: +1. Writing pack blobs to storage +2. Writing the index that references those packs + +During this window, the pack blobs are **unreferenced** - they exist in storage but no index points to them. If garbage collection runs during this window, it might delete these packs, corrupting the in-progress backup. + +This is the fundamental problem that sessions solve. + +--- + +## Sessions: The Write Protection Mechanism + +### What is a Session? + +A session is Kopia's mechanism for protecting in-progress writes from garbage collection. When a write operation begins, Kopia: + +1. Generates a unique **Session ID** +2. Writes a **Session Marker Blob** to storage +3. Embeds the Session ID in all pack blob names created during the session + +### Session Lifecycle + +``` +┌─────────────────────────────────────────────────────────────┐ +│ Session Lifecycle │ +├─────────────────────────────────────────────────────────────┤ +│ │ +│ 1. SESSION START (on first write) │ +│ ├─ Generate unique Session ID │ +│ ├─ Write Session Marker blob ONCE (prefix: "s") │ +│ │ └─ CheckpointTime set to NOW │ +│ └─ Session Marker contains: │ +│ - ID │ +│ - StartTime │ +│ - CheckpointTime (= StartTime, never updated!) │ +│ - User, Host │ +│ │ +│ 2. DATA WRITING │ +│ ├─ Pack blobs created with Session ID in name │ +│ │ Format: {prefix}{randomID}-{sessionID} │ +│ ├─ Packs are UNREFERENCED (not in any index yet) │ +│ └─ Session Marker is NOT updated during this phase │ +│ │ +│ 3. SESSION COMMIT (during Flush) │ +│ ├─ Write Index Blobs (references all packs) │ +│ ├─ DELETE Session Marker blobs (not update - delete!) │ +│ └─ Packs are now REFERENCED (protected by index) │ +│ │ +│ Note: After commit, next write starts a NEW session │ +│ │ +└─────────────────────────────────────────────────────────────┘ +``` + +### Pack Blob Naming Convention + +Pack blobs include the session ID in their name: + +``` +p{random64bit}-{sessionID} # Regular pack blob +q{random64bit}-{sessionID} # Special pack blob (for indexes, etc.) +``` + +This allows garbage collection to identify which session created each pack and whether that session is still active. + +### Session Marker Structure + +Session markers are stored as blobs with the `s` prefix and contain JSON metadata: + +```json +{ + "id": "abc123-epoch5", + "startTime": "2024-01-15T10:00:00Z", + "checkpointTime": "2024-01-15T10:00:00Z", + "username": "backup-user", + "hostname": "backup-server" +} +``` + +The `checkpointTime` field is critical - it indicates when the session was last known to be active. + +### Critical: Session Markers Are Immutable + +**The session marker is written ONCE when the session starts and is NEVER updated during the session's lifetime.** + +From `sessions.go`: + +```go +func (bm *WriteManager) writeSessionMarkerLocked(ctx context.Context) error { + cp := bm.currentSessionInfo + cp.CheckpointTime = bm.timeNow() // Set at write time, never updated + // ... write to storage ... +} + +// TODO(jkowalski): write this periodically when sessions span the duration of an upload. +``` + +This means: +- `CheckpointTime` equals the session start time +- If a session runs for hours without committing, the CheckpointTime becomes stale +- After `SessionExpirationAge` (96 hours), the session appears "expired" to maintenance + +**This is why checkpoints are essential**: They don't update the session marker - they **commit the current session and start a new one** with a fresh CheckpointTime. + +### How Sessions Protect Blobs + +During garbage collection, when evaluating whether to delete an unreferenced pack: + +```go +// From pack_gc.go +sid := content.SessionIDFromBlobID(bm.BlobID) +if s, ok := activeSessions[sid]; ok { + if age := cutoffTime.Sub(s.CheckpointTime); age < safety.SessionExpirationAge { + // PRESERVE - pack belongs to an active session + return nil + } +} +// DELETE - pack is truly orphaned +``` + +The pack is preserved if: +1. Its session ID matches an active session marker, AND +2. The session's CheckpointTime is within `SessionExpirationAge` + +--- + +## Checkpoints: Periodic Data Protection + +### What Are Kopia Checkpoints? + +Kopia checkpoints are periodic save points during long-running backup operations. They serve two purposes: + +1. **Progress Saving**: Allow resuming interrupted backups +2. **Session Renewal**: Commit current session's data and start a fresh session + +**Important distinction**: Checkpoints do NOT update the existing session marker. Instead, they: +1. Write index blobs (making current packs referenced/protected) +2. Delete the current session marker (commit) +3. Let the next write operation create a NEW session with a fresh CheckpointTime + +This is why checkpoints protect long-running backups despite session markers being immutable. + +### The 45-Minute Checkpoint Interval + +By default, Kopia creates a checkpoint every 45 minutes during snapshot uploads: + +```go +// From upload.go +const DefaultCheckpointInterval = 45 * time.Minute +``` + +This interval is enforced as a maximum - you cannot set a longer interval: + +```go +if u.CheckpointInterval > DefaultCheckpointInterval { + return nil, errors.Errorf("checkpoint interval cannot be greater than %v", DefaultCheckpointInterval) +} +``` + +### What Happens During a Checkpoint + +When a checkpoint triggers, the following sequence occurs: + +``` +┌─────────────────────────────────────────────────────────────┐ +│ Checkpoint Sequence │ +├─────────────────────────────────────────────────────────────┤ +│ │ +│ 1. CHECKPOINT TRIGGERED (every 45 minutes) │ +│ │ +│ 2. Flush() CALLED │ +│ ├─ finishAllPacksLocked() │ +│ │ └─ Complete any pending pack writes │ +│ │ │ +│ └─ flushPackIndexesLocked() │ +│ ├─ Write Index Blobs │ +│ │ └─ All packs now REFERENCED (protected!) │ +│ └─ commitSession() │ +│ └─ DELETE Session Marker blobs │ +│ (session S1 is now finished) │ +│ │ +│ 3. NEW SESSION STARTS (on next write) │ +│ └─ getOrStartSessionLocked() creates NEW session S2 │ +│ └─ NEW Session Marker written │ +│ └─ CheckpointTime = NOW (fresh timestamp!) │ +│ │ +└─────────────────────────────────────────────────────────────┘ +``` + +**Key insight**: The old session marker is deleted, not updated. Protection for the old session's packs now comes from the index, not the session marker. + +### Checkpoints Protect Long-Running Backups + +For a 2-week database backup: + +``` +Hour 0.00: Session S1 starts (CheckpointTime = Hour 0) + Packs P1-P100 written (UNREFERENCED, protected by S1 marker) + +Hour 0.75: CHECKPOINT + ├─ Index I1 written → P1-P100 now REFERENCED by index + ├─ Session S1 marker DELETED (committed) + └─ P1-P100 no longer need session protection (index protects them) + +Hour 0.76: First write after checkpoint + └─ Session S2 starts (CheckpointTime = Hour 0.76) + Packs P101-P200 written (UNREFERENCED, protected by S2 marker) + +Hour 1.50: CHECKPOINT + ├─ Index I2 written → P101-P200 REFERENCED + ├─ Session S2 marker DELETED + └─ Session S3 will start on next write +... +Hour 336: Final flush, backup complete +``` + +**Key insight**: Even though the backup takes 2 weeks, at any given time only the most recent 45 minutes of data is unreferenced. All earlier data has been indexed and is protected by the index, not by session markers. + +**The 96-hour SessionExpirationAge is a safety margin**, not a limit on backup duration. It protects against scenarios where checkpoints fail repeatedly. diff --git a/documentation/web/versioned_docs/version-0.0.18/developer/openshift_testing.md b/documentation/web/versioned_docs/version-0.0.18/developer/openshift_testing.md new file mode 100644 index 00000000..99428031 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/developer/openshift_testing.md @@ -0,0 +1,279 @@ +--- +sidebar_position: 4 +--- + +# OpenShift + +## Security Context Constraints + +OpenShift enforces +[Security Context Constraints](https://docs.redhat.com/en/documentation/openshift_container_platform/latest/html/authentication_and_authorization/managing-pod-security-policies) +(SCCs) to control the actions that pods can perform and the resources +they can access. Under the default **restricted** SCC, containers are +not allowed to run as a fixed user ID. Instead, OpenShift assigns an +arbitrary UID from a range that is unique to each namespace. + +As described in the OpenShift +[image creation guidelines](https://docs.redhat.com/en/documentation/openshift_container_platform/latest/html/images/creating-images#use-uid_create-images): + +> By default, OpenShift Container Platform runs containers using an +> arbitrarily assigned user ID. This provides additional security +> against processes escaping the container due to a container engine +> vulnerability and thereby achieving escalated permissions on the +> host node. + +This means that specifying a fixed `runAsUser`, `runAsGroup`, or +`fsGroup` in a pod's security context is rejected by the restricted +SCC unless the value falls within the namespace's allocated range. + +### How Klio handles this + +The Klio Operator detects OpenShift at startup by querying the +Kubernetes discovery API for the `securitycontextconstraints` +resource in the `security.openshift.io/v1` API group. + +When running on OpenShift: + +- **Server pods**: The operator omits `runAsUser`, `runAsGroup`, + and `fsGroup` from the pod security context, allowing the + restricted SCC to assign a UID from the namespace's range. +- **Plugin sidecar containers**: The operator omits `runAsUser` + and `runAsGroup` from the container security context for the + same reason. + +On vanilla Kubernetes, the operator continues to set explicit +UIDs (1000 for server pods, 26 for plugin sidecars) as before. + +No user configuration is required. The detection is automatic and +logged at startup: + +``` +INFO setup Cluster capabilities detected {"haveSecurityContextConstraints": true} +``` + +## Testing on OpenShift with OLM + +This guide explains how to install a test build of the Klio operator +on an OpenShift cluster using OLM (Operator Lifecycle Manager). + +:::note +This procedure is for development and testing only. It requires +a catalog image built from the branch under test. +::: + +## Prerequisites + +- An OpenShift cluster with `oc` CLI configured +- Cluster admin privileges +- cert-manager installed in the cluster (for TLS certificate + creation) +- A catalog image built with `task olm:catalog ENVIRONMENT=testing` + +The Klio operator, bundle, catalog and operand images are public +on `ghcr.io`, so no pull secret is needed. + +## 1. Apply the CatalogSource + +A CI run creates a catalog image, tagged with the branch name or the PR number: + +Examples: +``` +ghcr.io/cloudnative-pg/klio-operator-testing:main-catalog +ghcr.io/cloudnative-pg/klio-operator-testing:pr-1325-catalog +``` + +Create an `openshift_catalogsource.yaml` file pointing to the +catalog image built from your branch: + +```yaml +apiVersion: operators.coreos.com/v1alpha1 +kind: CatalogSource +metadata: + name: klio-catalog + namespace: openshift-marketplace +spec: + sourceType: grpc + image: +``` + +Then apply it: + +```bash +oc apply -f openshift_catalogsource.yaml +``` + +Wait for the catalog pod to become ready: + +```bash +oc get pods -n openshift-marketplace -w | grep klio +``` + +## 2. Subscribe to the operator + +Create an `openshift_subscription.yaml` file: + +```yaml +apiVersion: operators.coreos.com/v1alpha1 +kind: Subscription +metadata: + name: klio-operator + namespace: openshift-operators +spec: + channel: stable-v0 + name: klio-operator + source: klio-catalog + sourceNamespace: openshift-marketplace + installPlanApproval: Automatic +``` + +Apply it: + +```bash +oc apply -f openshift_subscription.yaml +``` + +OLM will create an `InstallPlan` and deploy the operator into +the `openshift-operators` namespace. + +:::note +The Klio sidecar (operand) image is baked into the operator +Deployment by the bundle as the `SIDECAR_IMAGE` environment +variable. It uses the same registry and tag as the operator +image by default, with the `klio` repository instead of +`klio-operator`. The Subscription can override it. +::: + +## 3. Create TLS certificates + +The Klio plugin requires two TLS secrets to establish mutual +TLS with the CNPG operator: + +- `klio-plugin-server-tls` — presented by the plugin gRPC server +- `klio-plugin-client-tls` — used by CNPG to authenticate to + the plugin + +They can be created manually, but the recommended method is to use cert-manager to +automatically generate and manage the certificates. + +Create a file `openshift_certificates.yaml` with the following +content, adjusting the `namespace` if the operator was installed +outside of `openshift-operators`: + +```yaml +--- +apiVersion: cert-manager.io/v1 +kind: Issuer +metadata: + name: klio-operator-selfsigned-issuer + namespace: openshift-operators +spec: + selfSigned: {} +--- +apiVersion: cert-manager.io/v1 +kind: Certificate +metadata: + name: klio-plugin-server + namespace: openshift-operators +spec: + secretName: klio-plugin-server-tls + dnsNames: + - klio-operator-plugin + usages: + - server auth + issuerRef: + name: klio-operator-selfsigned-issuer + kind: Issuer + group: cert-manager.io + duration: 2160h + renewBefore: 360h +--- +apiVersion: cert-manager.io/v1 +kind: Certificate +metadata: + name: klio-plugin-client + namespace: openshift-operators +spec: + secretName: klio-plugin-client-tls + commonName: klio-plugin-client + usages: + - client auth + issuerRef: + name: klio-operator-selfsigned-issuer + kind: Issuer + group: cert-manager.io + duration: 2160h + renewBefore: 360h +``` + +Apply it: + +```bash +oc apply -f openshift_certificates.yaml +``` + +Verify that the secrets have been created by cert-manager: + +```bash +oc get secrets -n openshift-operators \ + klio-plugin-server-tls klio-plugin-client-tls +``` + +## 4. Verify the installation + +The installation is complete when the CSV reaches the +`Succeeded` phase. You can check its status with: + +```bash +oc get csv -n openshift-operators -w +``` + +## Red Hat certification + +Klio's operator image and OLM bundle are validated against Red Hat's +[Preflight](https://github.com/redhat-openshift-ecosystem/openshift-preflight) +certification policies. Two checks cover the two artifacts: + +- **`check container`** — static policy checks on the operator image + (labels, layers, license, base image). It needs no cluster and runs + in the Dagger engine via `task olm:preflight-container`. +- **`check operator`** — installs the bundle through OLM into a live + OpenShift cluster and verifies it is deployable. Because it needs a + real OpenShift cluster (OLM and Security Context Constraints), it runs + via `task olm:preflight-operator`, against the CRC cluster the + OpenShift E2E job starts or against a local CRC. The bundle and + catalog images are multi-arch (`linux/amd64` and `linux/arm64`), so + the check runs natively on either architecture — including CRC on an + Apple Silicon Mac. + +:::note + +Both checks are currently **disabled in CI**. The operator and operand +images are built on Debian instead of Red Hat UBI, which the +`check container` base-image policy rejects, and `check operator` is +parked alongside it. The steps are commented out in +`.github/workflows/ci.yml` and `.github/workflows/openshift-e2e.yml`, +ready to be restored once a UBI-based image variant is built again — see +[issue #85](https://github.com/cloudnative-pg/klio/issues/85). Both +tasks still work when run manually, as described below. + +::: + +### Run `check operator` against CRC + +Point `EXTERNAL_KUBECONFIG` at an OpenShift (CRC) cluster and run the +task. The bundle and catalog (index) images must already be published +for the build under test — build them first with +`task olm:catalog ENVIRONMENT=testing` if needed; the certification runs +against the existing bundle rather than rebuilding it. + +```bash +export EXTERNAL_KUBECONFIG=/path/to/crc/kubeconfig +task olm:preflight-operator ENVIRONMENT=testing +``` + +preflight installs the operator through OLM, runs the operator policy, +and the Dagger `preflight` module evaluates the pass/fail verdict +in-engine. Raw artifacts are written to `operator/preflight-artifacts/`. + +The task never contacts Red Hat: the checks are always evaluated +locally. diff --git a/documentation/web/versioned_docs/version-0.0.18/developer/protocol.mdx b/documentation/web/versioned_docs/version-0.0.18/developer/protocol.mdx new file mode 100644 index 00000000..1e683f95 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/developer/protocol.mdx @@ -0,0 +1,9 @@ +--- +sidebar_position: 4 +title: Protocol Documentation +hide_title: true +--- + +import Protocol from './_protocol.md'; + + diff --git a/documentation/web/versioned_docs/version-0.0.18/developer/running-e2e-tests.md b/documentation/web/versioned_docs/version-0.0.18/developer/running-e2e-tests.md new file mode 100644 index 00000000..5c173203 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/developer/running-e2e-tests.md @@ -0,0 +1,333 @@ +--- +sidebar_position: 2 +--- + +# Running E2E tests + +This guide explains how to run the end-to-end (E2E) tests for the +Klio project. The E2E tests validate the integration between Klio +components and CloudNativePG in a real Kubernetes environment. + +## Prerequisites + +Before running E2E tests, ensure you have the following tools installed: + +- [Docker](https://docs.docker.com/get-docker/) - Container runtime +- [Kind](https://kind.sigs.k8s.io/docs/user/quick-start/) - Kubernetes in Docker +- [Task](https://taskfile.dev/installation/) - Task runner +- [Go](https://golang.org/doc/install) - Go programming language +- [kubectl](https://kubernetes.io/docs/tasks/tools/) - Kubernetes command-line tool + +## Required Setup + +:::info +Before running any E2E tests, you must first set up the CloudNativePG environment. +::: + +### 1. Clone CloudNativePG Repository + +```bash +git clone https://github.com/cloudnative-pg/cloudnative-pg.git +cd cloudnative-pg +``` + +### 2. Create Kind Cluster with CloudNativePG + +```bash +# Create, load, and deploy the CloudNativePG environment +hack/setup-cluster.sh create load deploy +cd .. +``` + +This script will: + +- Create a Kind cluster with the correct configuration +- Load the CloudNativePG operator images +- Deploy the CloudNativePG operator to the cluster + +## Test Environment Setup + +The E2E tests require a specific environment setup that includes: + +1. **Kind cluster** with CloudNativePG operator installed ✅ (completed above) +1. **Klio operator** deployed to the cluster +1. **cert-manager** for TLS certificate management + +### Automated Setup with Task + +After completing the required setup above, you can use the Task +runner to deploy Klio and run tests: + +```bash +# Set the Kind cluster name (must match the pattern used by CloudNativePG) +export KIND_CLUSTER_NAME=pg-operator-e2e-v1-33-1 + +# Run the complete E2E test suite +task integration:e2e +``` + +This command will: + +- Deploy cert-manager to the existing Kind cluster +- Build and deploy the Klio operator +- Run all E2E tests + +### Manual Setup + +If you prefer to set up the environment manually or need more +control, after completing the required CloudNativePG setup above: + +#### 1. Deploy Klio Operator + +```bash +# Set the cluster name (check with: kind get clusters) +export KIND_CLUSTER_NAME=pg-operator-e2e-v1-33-1 + +# Deploy the operator to the Kind cluster +task integration:deploy-to-kind +``` + +#### 2. Run E2E Tests + +```bash +# Run the tests +cd operator/test/e2e +go test -v ./... +``` + +## Test Structure + +The E2E tests are located in `operator/test/e2e/` and include: + +- **`main_test.go`** - Test setup, configuration, and feature + registration +- **`common_test.go`** - Shared test utilities and helpers +- **`backup_test.go`** - Backup functionality tests: + - `BackupFromPrimary`: backup from a single-instance cluster + - `BackupFromStandby`: backup from a standby in a multi-instance + cluster +- **`maintenance_test.go`** - Server-side post-backup maintenance on a + tier1-only deployment: verifies the backup queue consumer applies + tier1 WAL retention after a backup even when tier2 is not configured + (`Tier1ServerSideMaintenance`) +- **`recovery_test.go`** - Cluster recovery tests: + - `RecoverClusterFromBackupID`: recovery using a specific backup ID + - `RecoverClusterFromLatestBackup`: recovery from the latest backup + - `RecoverClusterFromPitr`: point-in-time recovery (tier1) + - `RecoverReplicaCluster`: replica cluster creation from backup +- **`tablespace_recovery_test.go`** - Recovery preserving PostgreSQL + tablespaces (`RecoverClusterWithTablespaces`) +- **`tier2_recovery_test.go`** - Recovery from tier2 S3 storage + (`RecoverClusterFromTier2`) +- **`tier2_pitr_test.go`** - Point-in-time recovery from tier2 storage + (`RecoverClusterFromTier2Pitr`) +- **`tier2_retention_test.go`** - Backup and WAL retention policy + enforcement in tier2 storage (`Tier2Retention`) +- **`wal_retention_test.go`** - WAL retention queue-awareness: verifies + server-side tier1 retention prunes old WALs only after they reach tier2, + driven by backup completion rather than a client command + (`WALRetentionQueueAwareness`) +- **`server_reconfig_test.go`** - Adding tier2 storage to an existing + tier1+queue server (`ServerTierReconfiguration`) +- **`pluginconfiguration_update_test.go`** - PluginConfiguration updates + and sidecar restart behavior (`PluginConfigurationUpdate`) +- **`pvc_resize_test.go`** - PVC resize for data, cache, and queue + volumes (`PVCResize`) +- **`otel_test.go`** - OpenTelemetry metrics and traces export: deploys + an OTEL Collector and verifies that backup lifecycle metrics and + traces are correctly exported via OTLP. After the success-path + assertions it deletes the Klio server and triggers a failing backup to + verify the `failure_category=repository_error` attribute on + `klio.plugin.backup.runs` (`OTELMetricsAndTraces`) +- **`operator_otel_test.go`** - Operator OpenTelemetry metrics: + patches the operator deployment with OTEL env vars and verifies + that controller-runtime metrics are bridged to an OTLP collector + (`OperatorOTELMetrics`, serial) + +## Test Configuration + +The e2e tests are configured through a YAML file located at +`operator/test/e2e/e2e-config.yaml`. The file is committed with +default values that match the local development environment. + +Available options: + +```yaml +# Klio server container image used in e2e tests. +serverImage: "registry.dev:5000/klio-testing:dev" + +# Directory where pod logs are streamed during the test run. +logDir: "e2e_cluster_logs" + +# Kubernetes storage class used for all PVC templates in the tests. +storageClass: "csi-hostpath-sc" + +# Namespace where the Klio operator runs. Defaults to cnpg-system (Helm/Kind); +# set to openshift-operators for the OLM-based OpenShift install. +operatorNamespace: "cnpg-system" + +# Label selector for identifying the Klio operator deployment. Must match +# the labels applied in the operator deployment manifest. +operatorAppLabel: "app.kubernetes.io/name=klio" + +# Name of the OLM Subscription managing the operator (in operatorNamespace). +# Set it on OpenShift so tests that change the operator environment patch the +# Subscription (which OLM propagates to the Deployment) rather than the +# Deployment, which OLM reverts. Leave empty for the Helm/Kind install. +operatorSubscription: "" +``` + +To use a custom configuration, either edit `e2e-config.yaml` directly +or point the `E2E_CONFIG_FILE` environment variable at an alternative +file: + +```bash +E2E_CONFIG_FILE=/path/to/my-config.yaml go test -v ./... +``` + +Configuration is resolved in two layers: built-in defaults, then the +config file. Pointing `E2E_CONFIG_FILE` at a personal file outside the +repository is the recommended way to keep local overrides (including +credentials) out of version control. + +## Log Collection + +During a test run, the test suite streams logs from all relevant pods in +the cluster using stern. Logs are written to a directory on the local +filesystem. The following components are covered: + +- Klio operator +- CNPG operator +- PostgreSQL instances +- RustFS server and init-job pods +- Klio server pods + +The log directory is controlled by the `logDir` key in the config file +(default: `e2e_cluster_logs/` relative to the test working directory). + +When running via `task integration:e2e`, logs are exported to +`e2e_cluster_logs/` in the repository root after the test run +completes. In CI, that directory is uploaded as a workflow artifact +named `e2e_cluster_logs` and is available for download from the +GitHub Actions run page. + +### Namespace object dumps on failure + +When a feature fails, its teardown writes a JSON snapshot of the +objects in the test namespace under `//`, one +`.json` file per resource kind (for example `PodList.json` and +`ClusterList.json`), before the namespace is deleted. It covers the Klio +and CloudNativePG custom resources together with the core workload +objects (Pods, PVCs, Jobs, Events, ServiceAccounts, EndpointSlices, +StatefulSets and Deployments), giving a point-in-time view of the +cluster state at the moment of failure. + +## Running on OpenShift + +The project includes an OpenShift-based e2e pipeline that validates +Klio on a single-node OpenShift (SNO) cluster. The cluster is +provisioned with [CRC](https://github.com/crc-org/crc-github-action) +(CodeReady Containers) using the `openshift` preset, which already +ships OLM, the operator catalogs, and a default StorageClass. The +pipeline then installs cert-manager, community CNPG, and the Klio +operator via OLM subscriptions and runs the same Go e2e suite as the +Kind path. + +### OpenShift prerequisites + +CRC runs the cluster inside a KVM virtual machine, so the runner must +expose `/dev/kvm` (nested virtualization). In addition to Go and Task, +the pipeline requires: + +- A pre-built Klio OLM catalog artifact + (`operator/klio-operator-catalog-source.yaml`) +- A Red Hat pull secret (`REDHAT_PULL`) used to start the CRC cluster. + This is the same secret, under the same name, that `cloudnative-pg` + uses to install its own OpenShift test clusters. + +The Klio operator, bundle, catalog and operand images are public on +`ghcr.io`, so no registry credentials are needed. + +:::note +The Red Hat pull secret is not available to forks, so the +`openshift-e2e` CI job is gated on the `OPENSHIFT_ENABLED` repository +variable and is skipped unless a repository opts in. To run it on a +fork, set the `OPENSHIFT_ENABLED` variable to `true` and add your own +`REDHAT_PULL` secret. Running the suite locally against your own CRC +cluster needs neither. +::: + +### Running locally + +Start a CRC `openshift` cluster yourself, then point the tasks at its +kubeconfig (CRC writes it to `~/.kube/config`): + +```bash +export EXTERNAL_KUBECONFIG="$HOME/.kube/config" + +# Generate the Klio OLM CatalogSource the deploy step requires (or +# download the klio-olm artifact from the ci.yml olm job instead). +task olm:catalog-source + +task integration:e2e-openshift +``` + +This command will: + +- Deploy cert-manager, CNPG, and Klio via OLM subscriptions +- Run the full e2e suite + +> **Storage note:** CRC's default StorageClass is hostPath-backed and +> cannot expand volumes, so `task integration:e2e-openshift` deploys +> the upstream `csi-driver-host-path` (the same driver the Kind path +> uses) via its `integration:setup-crc-storage` dependency and points +> `storageClass` at `csi-hostpath-sc`, which has `allowVolumeExpansion` +> enabled for the `PVCResize` feature. This runs the same way in CI and +> locally; override `STORAGE_CLASS` to use another expandable class. + +### CI + +The OpenShift e2e job does not run on regular pull requests. It runs +on pushes to `main`, on the `release-please--branches--main` release +PR, and on demand via the CI workflow's `workflow_dispatch` +(`run-openshift` input). When it runs, the `core` and `operator` jobs +are forced to build so their images are published to GHCR before the +suite pulls them; the deploy step also consumes the `klio-olm` +CatalogSource artifact produced by the always-on `olm` job. It is +defined in `.github/workflows/openshift-e2e.yml` and provisions the +cluster with the `crc-org/crc-github-action` action. + +## Environment Variables + +The following environment variables can be used to customize the +tasks execution: + +- `KIND_CLUSTER_NAME` - Name of the Kind cluster to use (required + for the Kind path) +- `E2E_CONFIG_FILE` - Path to a custom test configuration file + (default: `operator/test/e2e/e2e-config.yaml`) + +## Writing New Tests + +When adding new E2E tests: + +1. Follow the existing patterns in `backup_test.go` or + `recovery_test.go` +1. Use the machinery framework for test setup + 1. Add new feature types if needed. Everything in the framework must be + plugin-agnostic, and can be tested only through CloudNativePG APIs. +1. Register new features in `main_test.go` +1. Ensure proper cleanup of test resources + +## CI Integration + +The E2E tests are automatically run in CI when: + +- Changes are made to core components +- Changes are made to operator components +- Pull requests are submitted + +The CI environment automatically handles cluster setup and cleanup. +After the run, cluster logs are uploaded as a workflow artifact named +`e2e_cluster_logs` and can be downloaded from the GitHub Actions run +page regardless of whether the tests passed or failed. diff --git a/documentation/web/versioned_docs/version-0.0.18/user/_helm_chart_values.md b/documentation/web/versioned_docs/version-0.0.18/user/_helm_chart_values.md new file mode 100644 index 00000000..67797466 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/_helm_chart_values.md @@ -0,0 +1,42 @@ +| Key | Type | Default | Description | +|-----|------|---------|-------------| +| certmanager.clusterDomain | string | `"cluster.local"` | The DNS domain of the cluster | +| certmanager.createMetricsCertificate | bool | `true` | Create certificates for the metrics service. | +| certmanager.createPluginClientCertificate | bool | `true` | Create certificates for the plugin client. | +| certmanager.createPluginServerCertificate | bool | `true` | Create certificates for the plugin server. | +| certmanager.duration | string | `"2160h"` | The duration of the certificates. | +| certmanager.enable | bool | `true` | Enable cert-manager integration for certificate creation. | +| certmanager.renewBefore | string | `"360h"` | The renew before time for the certificates. | +| controllerManager.affinity | object | `{}` | Affinity rules for the operator deployment. | +| controllerManager.manager.args | list | `["--metrics-bind-address=:8443","--leader-elect","--health-probe-bind-address=:8081","--plugin-server-cert=/pluginServer/tls.crt","--plugin-server-key=/pluginServer/tls.key","--plugin-client-cert=/pluginClient/tls.crt","--plugin-server-address=:9090"]` | List of command line arguments to pass to the controller manager. | +| controllerManager.manager.containerSecurityContext | object | `{"allowPrivilegeEscalation":false,"capabilities":{"drop":["ALL"]}}` | The security context for the controller manager container. | +| controllerManager.manager.env | object | `{"SIDECAR_IMAGE":"ghcr.io/cloudnative-pg/klio:v0.0.18"}` | The environment variables to set in the controller manager container. Set OTEL_* variables here to enable OpenTelemetry export. See the OpenTelemetry Observability page in the documentation. | +| controllerManager.manager.image.pullPolicy | string | `"Always"` | The controller manager container imagePullPolicy. | +| controllerManager.manager.image.pullSecrets | list | `[]` | The list of imagePullSecrets. | +| controllerManager.manager.image.repository | string | `"ghcr.io/cloudnative-pg/klio-operator"` | The image to use for the controller manager container. | +| controllerManager.manager.image.tag | string | `"v0.0.18"` | The tag to use for the controller manager container image. | +| controllerManager.manager.livenessProbe | object | `{"httpGet":{"path":"/healthz","port":8081},"initialDelaySeconds":15,"periodSeconds":20}` | Liveness probe configuration. | +| controllerManager.manager.readinessProbe | object | `{"httpGet":{"path":"/readyz","port":8081},"initialDelaySeconds":5,"periodSeconds":10}` | Readiness probe configuration. | +| controllerManager.manager.resources | object | `{"limits":{"cpu":"500m","memory":"128Mi"},"requests":{"cpu":"10m","memory":"64Mi"}}` | The resources to allocate. | +| controllerManager.nodeSelector | object | `{}` | NodeSelector for the operator deployment. | +| controllerManager.podSecurityContext | object | `{"runAsNonRoot":true,"seccompProfile":{"type":"RuntimeDefault"}}` | The security context for the controller manager pod. | +| controllerManager.priorityClassName | string | `""` | Priority class name for the controller manager pod. | +| controllerManager.serviceAccount.annotations | object | `{}` | The annotations to add to the service account. | +| controllerManager.tolerations | list | `[]` | Tolerations for the operator deployment. | +| controllerManager.topologySpreadConstraints | list | `[]` | Topology Spread Constraints for the operator deployment. | +| fullnameOverride | string | `""` | Override the fully qualified name of the Helm Chart. | +| kubernetesClusterDomain | string | `"cluster.local"` | The domain for the Kubernetes cluster. | +| metricsService.enable | bool | `true` | Enable the metrics service for the controller manager. | +| metricsService.metricsServiceSecret | string | `"klio-metrics-server-cert"` | The name of the secret containing the TLS certificate for the metrics service. | +| metricsService.ports | list | `[{"name":"https","port":8443,"protocol":"TCP","targetPort":8443}]` | The port the metrics service will listen on. | +| metricsService.type | string | `"ClusterIP"` | Service type for the metrics service. | +| nameOverride | string | `"klio"` | Override the name of the Helm Chart. | +| plugin.clientSecret | string | `"klio-plugin-client-tls"` | The Client TLS certificate. | +| plugin.name | string | `"klio.cnpg.io"` | The name the plugin will use to register itself with the CNPG Operator. | +| plugin.port | int | `9090` | The port the plugin will listen on. It must match the "--plugin-server-address" argument. | +| plugin.serverSecret | string | `"klio-plugin-server-tls"` | The Server TLS certificate. | +| prometheus.enable | bool | `true` | To enable a ServiceMonitor to export metrics to Prometheus set true. | +| serviceAccount.annotations | object | `{}` | The annotations to add to the service account. | +| serviceAccount.automount | bool | `true` | Automount service account token. | +| serviceAccount.create | bool | `true` | Specifies whether a service account should be created. | +| serviceAccount.name | string | `""` | The name of the service account | diff --git a/documentation/web/versioned_docs/version-0.0.18/user/api/_category_.json b/documentation/web/versioned_docs/version-0.0.18/user/api/_category_.json new file mode 100644 index 00000000..c257999a --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/api/_category_.json @@ -0,0 +1,6 @@ +{ + "label": "API Reference", + "position": 99, + "collapsed": true, + "collapsible": true +} diff --git a/documentation/web/versioned_docs/version-0.0.18/user/api/_klio_api.md b/documentation/web/versioned_docs/version-0.0.18/user/api/_klio_api.md new file mode 100644 index 00000000..d666a314 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/api/_klio_api.md @@ -0,0 +1,440 @@ +## Packages +- [klio.cnpg.io/v1alpha1](#kliocnpgiov1alpha1) + + +## klio.cnpg.io/v1alpha1 + +Package v1alpha1 contains API Schema definitions for the klio v1alpha1 API group. + +### Resource Types +- [PluginConfiguration](#pluginconfiguration) +- [Server](#server) + + + +#### Cache + + + +Cache defines the configuration for the cache directory. + + + +_Appears in:_ +- [Tier1Configuration](#tier1configuration) +- [Tier2Configuration](#tier2configuration) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `pvcTemplate` _[PersistentVolumeClaimSpec](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#persistentvolumeclaimspec-v1-core)_ | | True | | | + + +#### Data + + + +Data defines the configuration for the data directory. + + + +_Appears in:_ +- [Tier1Configuration](#tier1configuration) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `pvcTemplate` _[PersistentVolumeClaimSpec](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#persistentvolumeclaimspec-v1-core)_ | Template to be used to generate the Persistent Volume Claim needed for the data folder,
containing base backups and WAL files. | True | | | + + +#### EmbeddedObjectMeta + + + +EmbeddedObjectMeta contains metadata for embedded objects. + + + +_Appears in:_ +- [PodTemplateSpec](#podtemplatespec) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `labels` _object (keys:string, values:string)_ | | | | Optional: \{\}
| +| `annotations` _object (keys:string, values:string)_ | | | | Optional: \{\}
| + + +#### FileReference + + + +FileReference specifies a file from a volume source. + + + +_Appears in:_ +- [FileSource](#filesource) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `volume` _[VolumeSource](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#volumesource-v1-core)_ | Volume is the volume source to mount. | True | | | +| `path` _string_ | Path is the file path within the mounted volume. | True | | | + + +#### FileSource + + + +FileSource specifies a source for a file. This wrapper allows future +alternatives to be added without breaking the API. + +_Validation:_ +- ExactlyOneOf: [fileReference] + +_Appears in:_ +- [Tier1Configuration](#tier1configuration) +- [Tier2Configuration](#tier2configuration) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `fileReference` _[FileReference](#filereference)_ | FileReference specifies a file from a volume source. | | | Optional: \{\}
| + + +#### ImageConfiguration + + + +ImageConfiguration contains the information needed to download +the Klio image. + + + +_Appears in:_ +- [ServerSpec](#serverspec) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `image` _string_ | Image is the image to be used for the Klio server | True | | | +| `imagePullPolicy` _[PullPolicy](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#pullpolicy-v1-core)_ | ImagePullPolicy defines the policy for pulling the image | | IfNotPresent | Optional: \{\}
| +| `imagePullSecrets` _[LocalObjectReference](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#localobjectreference-v1-core) array_ | ImagePullSecrets is an optional list of references to secrets in the same namespace to use for pulling any of the
images | | | Optional: \{\}
| + + +#### PluginConfiguration + + + +PluginConfiguration is the Schema for the client configuration API. + + + + + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `apiVersion` _string_ | `klio.cnpg.io/v1alpha1` | True | | | +| `kind` _string_ | `PluginConfiguration` | True | | | +| `metadata` _[ObjectMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#objectmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | True | | | +| `spec` _[PluginConfigurationSpec](#pluginconfigurationspec)_ | | True | | | +| `status` _[PluginConfigurationStatus](#pluginconfigurationstatus)_ | | | | Optional: \{\}
| + + +#### PluginConfigurationSpec + + + +PluginConfigurationSpec defines the desired state of client configuration. + + + +_Appears in:_ +- [PluginConfiguration](#pluginconfiguration) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `serverAddress` _string_ | ServerAddress is the address of the Klio server | True | | MinLength: 1
Required: \{\}
| +| `tier1` _[Tier1PluginConfiguration](#tier1pluginconfiguration)_ | Tier1 is the Tier 1 configuration | | | Optional: \{\}
| +| `tier2` _[Tier2PluginConfiguration](#tier2pluginconfiguration)_ | Tier2 is the Tier 2 configuration | | | Optional: \{\}
| +| `walPrefetch` _[WALPrefetchConfiguration](#walprefetchconfiguration)_ | WALPrefetch configures WAL prefetching behavior during recovery operations. | | | Optional: \{\}
| +| `clientSecretName` _string_ | ClientSecretName is the name of the secret containing the client credentials | True | | MinLength: 1
Required: \{\}
| +| `serverSecretName` _string_ | ServerSecretName is the name of the secret containing the server TLS certificate | True | | MinLength: 1
Required: \{\}
| +| `clusterName` _string_ | ClusterName is the name of the PostgreSQL cluster we are connecting to | True | | MinLength: 1
Required: \{\}
| +| `pprof` _boolean_ | Pprof enables the pprof endpoint for performance profiling | | | Optional: \{\}
| +| `mode` _[ServerMode](#servermode)_ | Mode selects the operation mode of the plugin. | True | standard | Enum: [standard read-only]
| +| `containers` _[Container](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#container-v1-core) array_ | Containers allows defining a list of containers that will be merged with the Klio sidecar containers.
This enables users to customize the sidecars with additional environment variables, volume mounts,
resource limits, and other container settings without polluting the PostgreSQL container environment.
Merge behavior:
- Containers are matched by name (klio-plugin, klio-restore)
- User customizations serve as the base
- Klio required values (name, args, CONTAINER_NAME env var) always override user values
- User-defined environment variables and volume mounts are preserved
- Template defaults are applied only for fields not set by the user or Klio | | | MaxItems: 2
Optional: \{\}
| + + +#### PluginConfigurationStatus + + + +PluginConfigurationStatus defines the observed state of ClientConfig. + + + +_Appears in:_ +- [PluginConfiguration](#pluginconfiguration) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `conditions` _[Condition](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#condition-v1-meta) array_ | Conditions represent the latest available observations of the
PluginConfiguration's state. | | | Optional: \{\}
| + + +#### PodTemplateSpec + + + +PodTemplateSpec describes the data a pod should have when created from a template. + + + +_Appears in:_ +- [ServerSpec](#serverspec) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `metadata` _[EmbeddedObjectMeta](#embeddedobjectmeta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | Optional: \{\}
| +| `spec` _[PodSpec](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#podspec-v1-core)_ | | | | Optional: \{\}
| + + +#### Queue + + + +Queue defines the configuration for the directory hosting the +task queue. + + + +_Appears in:_ +- [ServerSpec](#serverspec) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `pvcTemplate` _[PersistentVolumeClaimSpec](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#persistentvolumeclaimspec-v1-core)_ | PersistentVolumeClaimTemplate is used to generate the configuration for
the PVC hosting the work queue. | True | | | + + +#### RetentionPolicy + + + +RetentionPolicy defines how many backups we should keep. + + + +_Appears in:_ +- [Tier1PluginConfiguration](#tier1pluginconfiguration) +- [Tier2PluginConfiguration](#tier2pluginconfiguration) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `keepLatest` _integer_ | KeepLatest is the number of latest backups to keep
optional | True | | | +| `keepAnnual` _integer_ | KeepAnnual is the number of annual backups to keep
optional | True | | | +| `keepMonthly` _integer_ | KeepMonthly is the number of monthly backups to keep
optional | True | | | +| `keepWeekly` _integer_ | KeepWeekly is the number of weekly backups to keep
optional | True | | | +| `keepDaily` _integer_ | KeepDaily is the number of daily backups to keep
optional | True | | | +| `keepHourly` _integer_ | KeepHourly is the number of hourly backups to keep
optional | True | | | + + +#### S3Configuration + + + +S3Configuration is the configuration to a S3 defined tier 2. + + + +_Appears in:_ +- [Tier2Configuration](#tier2configuration) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `bucketName` _string_ | BucketName is the name of the bucket | True | | | +| `prefix` _string_ | Prefix is the path within the bucket under which all Klio objects
are stored, allowing a single bucket to be shared across multiple deployments. | | | Optional: \{\}
| +| `endpoint` _string_ | Endpoint is the endpoint to be used | | | Optional: \{\}
| +| `region` _string_ | Region is the region to be used | | | Optional: \{\}
| +| `accessKeyId` _[SecretKeySelector](https://pkg.go.dev/github.com/cloudnative-pg/machinery/pkg/api#SecretKeySelector)_ | The S3 access key ID | | | Optional: \{\}
| +| `secretAccessKey` _[SecretKeySelector](https://pkg.go.dev/github.com/cloudnative-pg/machinery/pkg/api#SecretKeySelector)_ | The S3 access key | | | Optional: \{\}
| +| `sessionToken` _[SecretKeySelector](https://pkg.go.dev/github.com/cloudnative-pg/machinery/pkg/api#SecretKeySelector)_ | The S3 session token | | | Optional: \{\}
| +| `customCaBundle` _[SecretKeySelector](https://pkg.go.dev/github.com/cloudnative-pg/machinery/pkg/api#SecretKeySelector)_ | A pointer to a custom CA bundle | | | Optional: \{\}
| + + +#### Server + + + +Server is the Schema for the servers API. + + + + + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `apiVersion` _string_ | `klio.cnpg.io/v1alpha1` | True | | | +| `kind` _string_ | `Server` | True | | | +| `metadata` _[ObjectMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#objectmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | True | | | +| `spec` _[ServerSpec](#serverspec)_ | | True | | | +| `status` _[ServerStatus](#serverstatus)_ | | | | Optional: \{\}
| + + +#### ServerMode + +_Underlying type:_ _string_ + +ServerMode defines the operation mode of the Server. + + + +_Appears in:_ +- [PluginConfigurationSpec](#pluginconfigurationspec) +- [ServerSpec](#serverspec) + +| Field | Description | +| --- | --- | +| `standard` | ModeStandard corresponds to server with standard read/write permissions.
| +| `read-only` | ModeReadOnly corresponds to a server with read-only permissions.
| + + +#### ServerSpec + + + +ServerSpec defines the desired state of Server. + + + +_Appears in:_ +- [Server](#server) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `image` _string_ | Image is the image to be used for the Klio server | True | | | +| `imagePullPolicy` _[PullPolicy](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#pullpolicy-v1-core)_ | ImagePullPolicy defines the policy for pulling the image | | IfNotPresent | Optional: \{\}
| +| `imagePullSecrets` _[LocalObjectReference](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.36/#localobjectreference-v1-core) array_ | ImagePullSecrets is an optional list of references to secrets in the same namespace to use for pulling any of the
images | | | Optional: \{\}
| +| `tlsSecretName` _string_ | TLSSecretName is the name of the Kubernetes secret containing the server-side certificate
to be used for the Klio server. | True | | | +| `caSecretName` _string_ | ClientCASecretName is the name of the Kubernetes secret containing the CA certificate
to be used by the Klio server to validate the users. | True | | | +| `mode` _[ServerMode](#servermode)_ | Mode selects the operation mode of the server. | True | standard | Enum: [standard read-only]
| +| `tier1` _[Tier1Configuration](#tier1configuration)_ | Tier1 is the Tier 1 configuration | True | | | +| `tier2` _[Tier2Configuration](#tier2configuration)_ | Tier2 is the Tier 2 configuration | True | | | +| `queue` _[Queue](#queue)_ | Queue is the configuration of the PVC that should host
the task queue. | | | Optional: \{\}
| +| `template` _[PodTemplateSpec](#podtemplatespec)_ | Template to override the default StatefulSet of the Klio server.
WARNING: Modifying this template may break the server functionality if not done carefully.
This field is primarily intended for advanced configuration such as telemetry setup.
Use at your own risk and ensure thorough testing before applying changes. | | | Optional: \{\}
| + + +#### ServerStatus + + + +ServerStatus defines the observed state of Server. + + + +_Appears in:_ +- [Server](#server) + + + +#### TLSConfiguration + + + +TLSConfiguration contains the information needed to configure +the PKI infrastructure of the Klio server. + + + +_Appears in:_ +- [ServerSpec](#serverspec) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `tlsSecretName` _string_ | TLSSecretName is the name of the Kubernetes secret containing the server-side certificate
to be used for the Klio server. | True | | | +| `caSecretName` _string_ | ClientCASecretName is the name of the Kubernetes secret containing the CA certificate
to be used by the Klio server to validate the users. | True | | | + + +#### Tier1Configuration + + + +Tier1Configuration is the tier 1 configuration. + + + +_Appears in:_ +- [ServerSpec](#serverspec) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `cache` _[Cache](#cache)_ | Cache is the configuration of the PVC that should be
used for the cache. | True | | | +| `data` _[Data](#data)_ | Data is the configuration of the PVC that should be used
for the base backups. | True | | | +| `encryptionKeyFile` _[FileSource](#filesource)_ | EncryptionKeyFile specifies the Age-encrypted encryption key file. | True | | ExactlyOneOf: [fileReference]
| +| `identityFile` _[FileSource](#filesource)_ | IdentityFile specifies the Age identity (private key) file used to
decrypt the encryption key. | True | | ExactlyOneOf: [fileReference]
| + + +#### Tier1PluginConfiguration + + + +Tier1PluginConfiguration configures tier1 backup and recovery settings. + + + +_Appears in:_ +- [PluginConfigurationSpec](#pluginconfigurationspec) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `retention` _[RetentionPolicy](#retentionpolicy)_ | RetentionPolicy defines how many backups we should keep | | | Optional: \{\}
| + + +#### Tier2Configuration + + + +Tier2Configuration is the tier 2 configuration. + + + +_Appears in:_ +- [ServerSpec](#serverspec) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `cache` _[Cache](#cache)_ | Cache is the configuration of the PVC that should be
used for the cache. | True | | | +| `s3` _[S3Configuration](#s3configuration)_ | S3 contains the configuration parameters for an S3-based tier 2. | True | | | +| `encryptionKeyFile` _[FileSource](#filesource)_ | EncryptionKeyFile specifies the Age-encrypted encryption key file. | True | | ExactlyOneOf: [fileReference]
| +| `identityFile` _[FileSource](#filesource)_ | IdentityFile specifies the Age identity (private key) file used to
decrypt the encryption key. | True | | ExactlyOneOf: [fileReference]
| + + +#### Tier2PluginConfiguration + + + +Tier2PluginConfiguration configures tier2 backup and recovery settings. + + + +_Appears in:_ +- [PluginConfigurationSpec](#pluginconfigurationspec) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `enableBackup` _boolean_ | EnableBackup controls whether WAL and base backups should be stored in tier2 | | | Optional: \{\}
| +| `enableRecovery` _boolean_ | EnableRecovery controls whether tier2 should be included in the recovery source list | | | Optional: \{\}
| +| `retention` _[RetentionPolicy](#retentionpolicy)_ | RetentionPolicy defines how many backups we should keep | | | Optional: \{\}
| + + +#### WALPrefetchConfiguration + + + +WALPrefetchConfiguration configures WAL prefetching during recovery. + + + +_Appears in:_ +- [PluginConfigurationSpec](#pluginconfigurationspec) + +| Field | Description | Required | Default | Validation | +| --- | --- | --- | --- | --- | +| `count` _integer_ | Count is the number of WAL files to prefetch ahead during recovery.
A value of 0 disables prefetching. | True | 2 | Maximum: 64
Minimum: 0
| +| `maxConcurrentDownloads` _integer_ | MaxConcurrentDownloads is the maximum number of concurrent WAL downloads. | True | 4 | Maximum: 64
Minimum: 1
| + + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/api/klio_api.mdx b/documentation/web/versioned_docs/version-0.0.18/user/api/klio_api.mdx new file mode 100644 index 00000000..c2cbf689 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/api/klio_api.mdx @@ -0,0 +1,5 @@ +import KlioAPI from './_klio_api.md'; + +# Klio API reference + + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/architectures.md b/documentation/web/versioned_docs/version-0.0.18/user/architectures.md new file mode 100644 index 00000000..c95b3fc9 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/architectures.md @@ -0,0 +1,205 @@ +--- +sidebar_position: 3 +--- + +# Architectures & Tiers + +Klio employs a multi-tiered architecture designed to balance performance, +resilience, and cost. This approach separates immediate, high-speed backup and +recovery operations from long-term archival and disaster recovery (DR) needs. +The architecture is built around two distinct storage tiers, each serving a +specific purpose in the data lifecycle. + +![Multi-tiered architecture overview](images/overview-multi-tiers.png) + +--- + +## Tier 1: Primary Storage (The Klio Server) + +**Tier 1** is the core operational tier, also referred to as the **Main Tier** +or **Klio Server**. It's designed for speed and provides immediate access to +all necessary backup artifacts for most recovery scenarios. + +This tier consists of a **local Persistent Volume (PV)** deployed by the +Klio Server. It can be located in the same namespace as the PostgreSQL cluster +or in a different one within the same Kubernetes cluster +(see the ["Tier 1 Architectures" section below](#tier-1-architectures)). + +Its purpose is to store the **WAL archive** and the **catalog of physical base +backups**. Its high-throughput, low-latency nature is optimized for several key +tasks: + +- Receiving a continuous stream of WAL files directly from the PostgreSQL + primary. +- Storing base backups created from the primary. +- Serving as the source for asynchronously replicating data to Tier 2. +- Managing retention policies for all tiers. + +### Tier 1 Architectures + +Klio supports several flexible deployment architectures for its Tier 1 storage. + +On the physical layer, it is recommended that both compute and, most +importantly, storage are separate from the PostgreSQL clusters. + +:::warning +Placing Tier 1 on the same nodes and storage as the PostgreSQL clusters +severely impacts the business continuity objectives of your organization. +::: + +On the logical layer, a **Klio Server** can reside in the same namespace as the +PostgreSQL cluster(s) it manages or in a separate, dedicated namespace. + +When choosing an architecture, it's important to consider +**security and tenancy**. +PostgreSQL clusters managed by a single Klio Server share the same master +encryption key. For this reason, it's recommended to use separate Klio Servers +for clusters that serve different tenants or have distinct security +requirements. + +#### Clusters and Klio Server in the Same Namespace + +The simplest deployment places the Klio Server in the same namespace as the +PostgreSQL cluster(s). + +This can be a **dedicated 1:1 mapping** (one Klio Server per cluster): + +![Cluster and Klio server in the same namespace](images/tier1-namespace-single.png) + +Or a **shared N:1 mapping** where one server manages all clusters in the +namespace. + +![Multiple clusters share a Klio server in the same namespace](images/tier1-namespace-multi.png) + +#### Clusters and Klio Server in Different Namespaces + +For greater isolation or centralized management, the Klio Server can be +deployed in a namespace separate from the PostgreSQL clusters it protects. + +The following diagram shows a PostgreSQL cluster being backed up by a Klio +Server in another namespace: + +![Cluster and Klio server in a different namespace](images/tier1-shared-single.png) + +This model also allows a central Klio Server to manage clusters that reside in +different namespaces, as shown below: + +![Multiple clusters share a Klio server in the same namespace](images/tier1-shared-multi.png) + +### Reserving Nodes for Klio Workloads + +For dedicated performance and resource isolation, you can reserve specific +worker nodes for Klio pods using Kubernetes taints and tolerations. + +1. **Taint the Node**: Apply a taint to the desired node. This prevents most + pods from being scheduled on it. + + ```sh + kubectl taint node node-role.kubernetes.io/klio=:NoSchedule + ``` + +1. **Add Toleration to Klio Server**: Add the corresponding toleration to your + Klio `Server` resource, adding it to `.spec.template`. + This allows the Klio Server to be scheduled on the tainted node. + + ```yaml + # In your Server resource definition + spec: + template: + spec: + containers: [] + tolerations: + - key: "node-role.kubernetes.io/klio" + operator: "Exists" + effect: "NoSchedule" + ``` + +--- + +## Tier 2: Secondary Storage (Object Storage) + +**Tier 2** provides durable, long-term storage for robust disaster recovery +(DR) strategies. It's physically and logically separate from the primary +Kubernetes cluster and typically consists of an external object storage system, +such as Amazon S3, Google Cloud Storage, or Azure Blob Storage. +Storing backups off-site ensures **geographical redundancy**, protecting data +against a full cluster or site failure. + +Klio asynchronously relays both base backups and WAL files from Tier 1 to +Tier 2. This decoupling ensures that primary backup and recovery operations in +Tier 1 are not directly affected by the latency or availability of the remote +object storage. + +Additionally, Tier 2 can serve as a read-only fallback source. In a distributed +CloudNativePG topology, this allows a Klio server at a secondary site to use +the shared Tier 2 storage to bootstrap a new cluster, enhancing DR +capabilities. + +### Snapshot Pinning + +When Tier 2 is enabled, Klio automatically pins snapshots in Tier 1 with a +`klio.io/tier2` pin. This mechanism prevents retention policies from +automatically deleting snapshots before they have been successfully migrated +to Tier 2. + +The pinning workflow operates as follows: + +1. When a backup is created with Tier 2 enabled, all snapshot components + (tablespaces, PGDATA, control file, and metadata) are tagged with the + `klio.io/tier2` pin. +1. The pin protects the snapshot from being removed by retention policy + enforcement, even if it would otherwise be eligible for deletion. +1. After the snapshot is successfully migrated to Tier 2, Klio removes the + pin, allowing normal retention policy management to resume. + +This ensures data integrity during the asynchronous migration process and +guarantees that no backup is lost due to retention policies running before +migration completes. + +### Restoring from Tier 2 + +Tier 2 is only used as a restore source, for both base backups and WAL files, +when `enableRecovery` is set to `true`. Enabling Tier 2 for backup alone +(`enableBackup: true`) does not opt a cluster into restoring from Tier 2. + +When `enableRecovery` is `true`, a backup requested for restore is first +looked up in Tier 1. If it is not found in Tier 1, Klio automatically checks +Tier 2. This fallback mechanism ensures that backups that have been migrated +to Tier 2 are still accessible for restore operations. + +When Tier 2 recovery is enabled and a backup exists in both tiers, Tier 1 +takes precedence as restore from it will be faster. + +### Read-Only Server Mode + +The Klio server supports **read-only mode** (`mode: read-only`), which serves +backups and WAL files from Tier 2 object storage without accepting write +operations. This mode is designed for disaster recovery scenarios where you +need restore capabilities without the cost of local storage or the risk of +accepting new backups. + +Read-only servers are particularly useful for CloudNativePG +[Replica Clusters](https://cloudnative-pg.io/docs/current/replica_cluster/) in +secondary regions or datacenters. Multiple read-only servers can restore from +a single S3 bucket populated by one primary server. + +In read-only mode, all read and restore operations from Tier 2 function +normally, while write operations (backup creation, WAL streaming, retention +policies) are rejected. + +See the [Configuring Read-Only Mode](klio_server.md#read-only-mode) +section for configuration examples. + +--- + +## Planning Your Backup Strategy + +When planning your backup strategy with Klio, **Tier 1 is the most critical +layer** to define architecturally. You have several options, ranging from +running Klio servers on any worker node using your cluster's primary storage +solution, to dedicating a single worker node with local storage for a +centralized Klio server. + +**Tier 2** is often determined by your organization's infrastructure teams, who +have likely already selected one or more standard object storage solutions for +long-term archival. diff --git a/documentation/web/versioned_docs/version-0.0.18/user/backup_and_restore.md b/documentation/web/versioned_docs/version-0.0.18/user/backup_and_restore.md new file mode 100644 index 00000000..8943ce33 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/backup_and_restore.md @@ -0,0 +1,303 @@ +--- +sidebar_position: 7 +--- + +# Backup and Restore + +This guide explains how to take backups of PostgreSQL clusters managed by +CloudNativePG and restore them using Klio. + +## Overview + +Klio follows PostgreSQL's native physical backup and recovery mechanisms, +leveraging CloudNativePG's backup and restore capabilities through its +[`Backup` resource](https://cloudnative-pg.io/documentation/current/cloudnative-pg.v1/#postgresql-cnpg-io-v1-Backup) +and +[`ScheduledBackup` resource](https://cloudnative-pg.io/documentation/current/cloudnative-pg.v1/#postgresql-cnpg-io-v1-ScheduledBackup). + +A working **online backup** is composed of: + +- A **physical base backup**: A filesystem copy of the PostgreSQL data directory. +- A set of **WAL (Write-Ahead Log) files**: Continuous logs of all changes made + to the database during the entire period of the base backup. + +:::tip +It is recommended to periodically test backup restores to ensure correct +recovery procedures. +::: + +## Prerequisites + +Before performing backup and restore operations, ensure you have: + +- A running [Klio server](./klio_server.md) with proper configuration +- A PostgreSQL cluster configured with the [Klio plugin](./plugin_configuration.md) + +## Taking a Backup + +With the Klio plugin configured, you can take on-demand backups using +CloudNativePG's [`Backup` resource](https://cloudnative-pg.io/documentation/current/cloudnative-pg.v1/#postgresql-cnpg-io-v1-Backup) +or the [Kubectl plugin](https://cloudnative-pg.io/documentation/current/kubectl-plugin/#requesting-a-new-physical-backup) +for CNPG. + +### Create a Backup + +You can trigger a new backup by creating a `Backup` resource. + +```yaml +apiVersion: postgresql.cnpg.io/v1 +kind: Backup +metadata: + name: my-cluster-backup-20251027 + namespace: default +spec: + method: plugin + target: primary + cluster: + name: my-cluster + pluginConfiguration: + name: klio.cnpg.io +``` + +Apply the manifest: + +```bash +kubectl apply -f backup.yaml +``` + +Alternatively, you can request a backup directly using the + [`kubectl cnpg` plugin](https://cloudnative-pg.io/documentation/current/kubectl-plugin/#requesting-a-new-physical-backup): + +```bash +kubectl cnpg backup my-cluster \ + --method plugin \ + --plugin-name klio.cnpg.io \ + --backup-target primary +``` + +If you don’t specify the `--backup-name` option, the `cnpg backup` command +automatically generates one using the format `-`, +which is suitable in most cases. + +For a complete list of available options, run: + +```bash +kubectl cnpg backup --help +``` + +### Monitor Backup Progress + +Check the backup status: + +```bash +# Watch the backup status +kubectl get backup my-cluster-backup-20251027 -w + +# Get detailed backup information +kubectl describe backup my-cluster-backup-20251027 +``` + +A successful backup will show: + +``` +NAME AGE CLUSTER METHOD PHASE ERROR +my-cluster-backup-20251027 2m my-cluster plugin Completed +``` + +### Scheduled Backups + +You can schedule automatic backups using CloudNativePG's +[`ScheduledBackup` resource](https://cloudnative-pg.io/documentation/current/cloudnative-pg.v1/#postgresql-cnpg-io-v1-ScheduledBackup). + +```yaml +apiVersion: postgresql.cnpg.io/v1 +kind: ScheduledBackup +metadata: + name: my-cluster-daily-backup + namespace: default +spec: + # Cron schedule: daily at 2:00 AM + schedule: "0 0 2 * * *" + method: plugin + target: primary + cluster: + name: my-cluster + pluginConfiguration: + name: klio.cnpg.io +``` + +Apply the scheduled backup: + +```bash +kubectl apply -f scheduled-backup.yaml +``` + +## Backup Retention and Maintenance + +Klio automatically manages backup retention based on the +[retention policies](plugin_configuration.md#retention-policies) defined in the +`PluginConfiguration` referred by the `Cluster`. + +:::warning +Deleting a `Backup` resource through `kubectl` only removes the Kubernetes +object. The actual backup data in the Klio server will be retained according to +the retention policy. +::: + +## Finding Your backupID for Recovery + +To restore a specific backup, you need its backupID, otherwise Klio will +choose the latest one autonomously. +You can list all available, completed Backup resources using kubectl: + +```bash +kubectl get backups -n +``` + +Once you identify the backup you want to use, you can identify its backupID + +```bash +kubectl get backup -n -o jsonpath='{.status.backupId}' +``` + +## Restoring from a Backup + +Klio supports restoring PostgreSQL clusters from backups using CloudNativePG's +recovery mechanism. Unlike traditional in-place recovery, Klio follows +CloudNativePG's approach of **bootstrapping a new cluster** from a backup, +which ensures data integrity and allows for flexible recovery scenarios. + +### How Recovery Works + +Klio integrates with CloudNativePG's recovery process by performing the +following actions during a restore: + +1. **Restores the base backup**: Copies the physical backup data to the new + cluster's data directory. Uses `klio restore` command under the hood. +1. **Restores WAL files**: Klio is configured to retrieve the WAL files + required for PostgreSQL recovery as needed. + Uses `klio get-wal` command under the hood. + +The execution of these commands is driven by CloudNativePG's recovery +mechanism, which ensures that the PostgreSQL server starts correctly after +the restore. + +A restored cluster operates independently of the original cluster. By default, +it will **not** perform backups unless you explicitly configure the Klio plugin +for backup operations in the new cluster's specification. + +### Full Restore + +To restore from a backup, create a new `Cluster` resource with a +`bootstrap.recovery` section that references the Klio plugin: + +```yaml +apiVersion: postgresql.cnpg.io/v1 +kind: Cluster +metadata: + name: my-restored-cluster + namespace: default +spec: + instances: 3 + + # Bootstrap from a Klio backup + bootstrap: + recovery: + source: source + # OPTIONAL: Specify the backup to restore from + recoveryTarget: + backupID: my-cluster-backup-YYYYMMDDHHMMSS + + # Reference the Klio plugin configuration + externalClusters: + - name: source + plugin: + name: klio.cnpg.io + parameters: + pluginConfigurationRef: my-restore-config + + storage: + size: 10Gi +``` + +:::note +Klio will choose the latest backup available in case the `backupID` field is +omitted. +::: + +Create a corresponding `PluginConfiguration` that specifies which backup to +restore: + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: PluginConfiguration +metadata: + name: my-restore-config + namespace: default +spec: + # Connection details + serverAddress: klio-server.default + clientSecretName: my-client-credentials + serverSecretName: klio-server-tls + + # Required: the name of the original cluster that was backed up + clusterName: my-cluster +``` + +The client credentials secret (`my-client-credentials`) should contain the +necessary authentication information to access the Klio server, as described +in the [Klio plugin configuration guide](./plugin_configuration.md#client-credentials-secret). + +:::note +The `clusterName` field in the `PluginConfiguration` and the `commonName` +of the certificate should match the name of the **original cluster** that +was backed up, not the name of the new restored cluster. +::: + +Apply both resources: + +```bash +kubectl apply -f restore-config.yaml +kubectl apply -f restored-cluster.yaml +``` + +### Point-in-Time Recovery (PITR) + +Klio supports Point-in-Time Recovery, allowing you to restore your database +to a specific moment in time rather than the latest available state. This is +useful for recovering from accidental data deletion or corruption. + +The process involves specifying a recovery target in the `Cluster` resource. +The available recovery targets are described in the +[CloudNativePG documentation](https://cloudnative-pg.io/documentation/current/recovery/#recovery-targets). + +#### Example: recover to a `targetTime` + +Restore to a specific timestamp: + +```yaml +apiVersion: postgresql.cnpg.io/v1 +kind: Cluster +metadata: + name: my-pitr-cluster +spec: + bootstrap: + recovery: + source: source + # Recover to a specific point in time + recoveryTarget: + targetTime: "2025-11-06 15:00:00.0000+00" + # other cluster spec fields... +``` + +:::info +The target of a point in time recovery must fall between the time the base +backup was completed and the time of the latest transaction recorded in the +available WAL files. +::: + +:::note +During the Point in Time Recovery, if `targetTime` or `targetLSN` are specified, +Klio will automatically choose the closest backup for the PITR, if not defined +with the `backupID` field. +::: diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/_category_.json b/documentation/web/versioned_docs/version-0.0.18/user/cli/_category_.json new file mode 100644 index 00000000..4686f208 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/_category_.json @@ -0,0 +1,10 @@ +{ + "label": "CLI Reference", + "position": 150, + "collapsed": true, + "collapsible": true, + "link": { + "type": "generated-index", + "description": "This CLI is primarily invoked internally by the Klio Operator, running inside Klio Server pods and as sidecar containers in PostgreSQL instance pods managed by CloudNativePG. Most of its commands are not generally meant to be run directly by end users." + } +} diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio.md new file mode 100644 index 00000000..73d82172 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio.md @@ -0,0 +1,49 @@ +--- +title: klio +--- + +## klio + +PostgreSQL Backup & Recovery for CloudNativePG + +### Synopsis + +Klio is a backup and recovery engine for PostgreSQL clusters managed by CloudNativePG. + +This CLI is primarily invoked internally by the Klio Operator: it runs inside +Klio Server pods and as sidecar containers in the PostgreSQL instance pods +managed by CloudNativePG. Most of its commands are not generally meant to be +run directly by end users. + +### Options + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + -h, --help help for klio + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + -t, --toggle Help message for toggle + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio admin](klio_admin.md) - Server administration commands +* [klio backup](klio_backup.md) - Manage physical backups +* [klio get-metadata](klio_get-metadata.md) - Get the metadata of a cluster from the target Klio server +* [klio get-wal](klio_get-wal.md) - Get a WAL from the target Klio server +* [klio reset-lsn](klio_reset-lsn.md) - Reset the replication status to the latest flush LSN +* [klio restore](klio_restore.md) - Restore a PostgreSQL cluster from a Klio server +* [klio retention](klio_retention.md) - Manage the retention policy +* [klio send-wal](klio_send-wal.md) - Upload the cluster's WALs to the target Klio server +* [klio server](klio_server.md) - Starts and manage a Klio server +* [klio wal-player](klio_wal-player.md) - WAL Player Commands + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin.md new file mode 100644 index 00000000..f3e1d229 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin.md @@ -0,0 +1,39 @@ +--- +title: klio admin +--- + +## klio admin + +Server administration commands + +### Options + +``` + -h, --help help for admin +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio](klio.md) - PostgreSQL Backup & Recovery for CloudNativePG +* [klio admin delete-backup](klio_admin_delete-backup.md) - Delete a backup from the Klio server +* [klio admin list-backups](klio_admin_list-backups.md) - List the backups available in the Klio server +* [klio admin queue](klio_admin_queue.md) - Manage the queue tasks +* [klio admin refresh](klio_admin_refresh.md) - Refresh the Kopia cache and policies + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_delete-backup.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_delete-backup.md new file mode 100644 index 00000000..de0bcef8 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_delete-backup.md @@ -0,0 +1,43 @@ +--- +title: klio admin delete-backup +--- + +## klio admin delete-backup + +Delete a backup from the Klio server + +``` +klio admin delete-backup [backupName] [flags] +``` + +### Options + +``` + --cluster string The name of the cluster that owns the backup + -h, --help help for delete-backup + --socket-path string Unix socket used by the administration server (default "/tmp/.klio-admin") + --tier1 Delete the backup from tier1 (local cache) + --tier2 Delete the backup from tier2 (object storage) +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio admin](klio_admin.md) - Server administration commands + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_list-backups.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_list-backups.md new file mode 100644 index 00000000..0bf1453c --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_list-backups.md @@ -0,0 +1,40 @@ +--- +title: klio admin list-backups +--- + +## klio admin list-backups + +List the backups available in the Klio server + +``` +klio admin list-backups [flags] +``` + +### Options + +``` + -h, --help help for list-backups + --socket-path string Unix socket used by the administration server (default "/tmp/.klio-admin") +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio admin](klio_admin.md) - Server administration commands + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue.md new file mode 100644 index 00000000..79bf776d --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue.md @@ -0,0 +1,40 @@ +--- +title: klio admin queue +--- + +## klio admin queue + +Manage the queue tasks + +### Options + +``` + -h, --help help for queue + --json Output in JSON format + --socket-path string Unix socket used by the administration server (default "/tmp/.klio-admin") +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio admin](klio_admin.md) - Server administration commands +* [klio admin queue backup](klio_admin_queue_backup.md) - Manage the queue backup tasks +* [klio admin queue status](klio_admin_queue_status.md) - Show the status of the task queue (pending backups and pending WALs) +* [klio admin queue wal](klio_admin_queue_wal.md) - Manage the queue WAL tasks + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_backup.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_backup.md new file mode 100644 index 00000000..a1937143 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_backup.md @@ -0,0 +1,38 @@ +--- +title: klio admin queue backup +--- + +## klio admin queue backup + +Manage the queue backup tasks + +### Options + +``` + -h, --help help for backup +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --json Output in JSON format + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --socket-path string Unix socket used by the administration server (default "/tmp/.klio-admin") + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio admin queue](klio_admin_queue.md) - Manage the queue tasks +* [klio admin queue backup list-failed](klio_admin_queue_backup_list-failed.md) - List failed backup tasks in the queue + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_backup_list-failed.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_backup_list-failed.md new file mode 100644 index 00000000..bb203d1e --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_backup_list-failed.md @@ -0,0 +1,42 @@ +--- +title: klio admin queue backup list-failed +--- + +## klio admin queue backup list-failed + +List failed backup tasks in the queue + +``` +klio admin queue backup list-failed [flags] +``` + +### Options + +``` + --cluster-name string Cluster name to filter failed backup tasks (optional) + -h, --help help for list-failed +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --json Output in JSON format + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --socket-path string Unix socket used by the administration server (default "/tmp/.klio-admin") + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio admin queue backup](klio_admin_queue_backup.md) - Manage the queue backup tasks + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_status.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_status.md new file mode 100644 index 00000000..6d9e8d0c --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_status.md @@ -0,0 +1,41 @@ +--- +title: klio admin queue status +--- + +## klio admin queue status + +Show the status of the task queue (pending backups and pending WALs) + +``` +klio admin queue status [flags] +``` + +### Options + +``` + -h, --help help for status +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --json Output in JSON format + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --socket-path string Unix socket used by the administration server (default "/tmp/.klio-admin") + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio admin queue](klio_admin_queue.md) - Manage the queue tasks + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_wal.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_wal.md new file mode 100644 index 00000000..562a303b --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_wal.md @@ -0,0 +1,38 @@ +--- +title: klio admin queue wal +--- + +## klio admin queue wal + +Manage the queue WAL tasks + +### Options + +``` + -h, --help help for wal +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --json Output in JSON format + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --socket-path string Unix socket used by the administration server (default "/tmp/.klio-admin") + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio admin queue](klio_admin_queue.md) - Manage the queue tasks +* [klio admin queue wal list-failed](klio_admin_queue_wal_list-failed.md) - List failed WAL tasks in the queue + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_wal_list-failed.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_wal_list-failed.md new file mode 100644 index 00000000..a2f0c235 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_queue_wal_list-failed.md @@ -0,0 +1,42 @@ +--- +title: klio admin queue wal list-failed +--- + +## klio admin queue wal list-failed + +List failed WAL tasks in the queue + +``` +klio admin queue wal list-failed [flags] +``` + +### Options + +``` + --cluster-name string Cluster name to filter failed WAL tasks (optional) + -h, --help help for list-failed +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --json Output in JSON format + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --socket-path string Unix socket used by the administration server (default "/tmp/.klio-admin") + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio admin queue wal](klio_admin_queue_wal.md) - Manage the queue WAL tasks + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_refresh.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_refresh.md new file mode 100644 index 00000000..ddc9a57c --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_admin_refresh.md @@ -0,0 +1,40 @@ +--- +title: klio admin refresh +--- + +## klio admin refresh + +Refresh the Kopia cache and policies + +``` +klio admin refresh [flags] +``` + +### Options + +``` + -h, --help help for refresh + --socket-path string Unix socket used by the administration server (default "/tmp/.klio-admin") +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio admin](klio_admin.md) - Server administration commands + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup.md new file mode 100644 index 00000000..5872da43 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup.md @@ -0,0 +1,40 @@ +--- +title: klio backup +--- + +## klio backup + +Manage physical backups + +### Options + +``` + -h, --help help for backup +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio](klio.md) - PostgreSQL Backup & Recovery for CloudNativePG +* [klio backup delete](klio_backup_delete.md) - Deletes the metadata with the provided name +* [klio backup get-metadata](klio_backup_get-metadata.md) - Gets the metadata of the backup with the provided name +* [klio backup list](klio_backup_list.md) - Gets the metadata of all backups +* [klio backup run](klio_backup_run.md) - Backup the PostgreSQL cluster to the opened Klio server +* [klio backup verify](klio_backup_verify.md) - Verify the integrity of backups + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_delete.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_delete.md new file mode 100644 index 00000000..478a65d5 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_delete.md @@ -0,0 +1,40 @@ +--- +title: klio backup delete +--- + +## klio backup delete + +Deletes the metadata with the provided name + +``` +klio backup delete [backupName] [flags] +``` + +### Options + +``` + -h, --help help for delete + -n, --name string The backup name +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio backup](klio_backup.md) - Manage physical backups + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_get-metadata.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_get-metadata.md new file mode 100644 index 00000000..a89d4667 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_get-metadata.md @@ -0,0 +1,40 @@ +--- +title: klio backup get-metadata +--- + +## klio backup get-metadata + +Gets the metadata of the backup with the provided name + +``` +klio backup get-metadata [backupName] [flags] +``` + +### Options + +``` + -h, --help help for get-metadata + -n, --name string The backup name +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio backup](klio_backup.md) - Manage physical backups + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_list.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_list.md new file mode 100644 index 00000000..0e768aa0 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_list.md @@ -0,0 +1,39 @@ +--- +title: klio backup list +--- + +## klio backup list + +Gets the metadata of all backups + +``` +klio backup list [flags] +``` + +### Options + +``` + -h, --help help for list +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio backup](klio_backup.md) - Manage physical backups + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_run.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_run.md new file mode 100644 index 00000000..86896030 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_run.md @@ -0,0 +1,42 @@ +--- +title: klio backup run +--- + +## klio backup run + +Backup the PostgreSQL cluster to the opened Klio server + +``` +klio backup run [flags] +``` + +### Options + +``` + --enable-tier2-backup When enabled, require the backup to be sent to tier2 + -h, --help help for run + -n, --name string The backup name + --wait-for-wals When enabled, wait until all the required WAL files have been archived in tier1 before declaring the backup completed. (default true) +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio backup](klio_backup.md) - Manage physical backups + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_verify.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_verify.md new file mode 100644 index 00000000..b40eda9f --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_backup_verify.md @@ -0,0 +1,49 @@ +--- +title: klio backup verify +--- + +## klio backup verify + +Verify the integrity of backups + +### Synopsis + +Verify the integrity of backups in the repository. + +By default, verifies only the backup names passed as arguments. +Use --all to verify all backups in the repository. +Use --tiers to select which tiers to verify (default: both). + +``` +klio backup verify [backup-names...] [flags] +``` + +### Options + +``` + --all Verify all backups instead of specific names + -h, --help help for verify + --tiers string Tiers to verify (tier1, tier2, or tier1,tier2) (default "tier1,tier2") +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio backup](klio_backup.md) - Manage physical backups + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_get-metadata.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_get-metadata.md new file mode 100644 index 00000000..9e1187cb --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_get-metadata.md @@ -0,0 +1,39 @@ +--- +title: klio get-metadata +--- + +## klio get-metadata + +Get the metadata of a cluster from the target Klio server + +``` +klio get-metadata [flags] +``` + +### Options + +``` + -h, --help help for get-metadata +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio](klio.md) - PostgreSQL Backup & Recovery for CloudNativePG + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_get-wal.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_get-wal.md new file mode 100644 index 00000000..14839bd8 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_get-wal.md @@ -0,0 +1,41 @@ +--- +title: klio get-wal +--- + +## klio get-wal + +Get a WAL from the target Klio server + +``` +klio get-wal [wal-name] [target-file] [flags] +``` + +### Options + +``` + -h, --help help for get-wal + --partial Use a partial WAL file if a the completed WAL file is not present. Defaults to false + --tier string The tier where we should look for WAL. Accepted values: 'tier1' and 'tier2' (default "tier1") +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio](klio.md) - PostgreSQL Backup & Recovery for CloudNativePG + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_reset-lsn.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_reset-lsn.md new file mode 100644 index 00000000..652d6ae4 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_reset-lsn.md @@ -0,0 +1,39 @@ +--- +title: klio reset-lsn +--- + +## klio reset-lsn + +Reset the replication status to the latest flush LSN + +``` +klio reset-lsn [flags] +``` + +### Options + +``` + -h, --help help for reset-lsn +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio](klio.md) - PostgreSQL Backup & Recovery for CloudNativePG + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_restore.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_restore.md new file mode 100644 index 00000000..abed4cd1 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_restore.md @@ -0,0 +1,42 @@ +--- +title: klio restore +--- + +## klio restore + +Restore a PostgreSQL cluster from a Klio server + +``` +klio restore [destination] [flags] +``` + +### Options + +``` + --backup-id string The ID of the backup to be restored. Defaults to the latest backup found for the cluster + -h, --help help for restore + --tablespaces strings A comma-separated list of tablespace_name:tablespace_path to customize the restore location of a tablespace. + --target-time string If specified, Klio will recover from the most recent backup that was taken before the time specified. This allows to configure PostgreSQL to do a point-in-time recovery with the specified target time. +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio](klio.md) - PostgreSQL Backup & Recovery for CloudNativePG + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_retention.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_retention.md new file mode 100644 index 00000000..d0116087 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_retention.md @@ -0,0 +1,37 @@ +--- +title: klio retention +--- + +## klio retention + +Manage the retention policy + +### Options + +``` + -h, --help help for retention +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio](klio.md) - PostgreSQL Backup & Recovery for CloudNativePG +* [klio retention get](klio_retention_get.md) - Gets the currently applied retention policy +* [klio retention set](klio_retention_set.md) - Sets the currently applied retention policy + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_retention_get.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_retention_get.md new file mode 100644 index 00000000..4cbba32c --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_retention_get.md @@ -0,0 +1,39 @@ +--- +title: klio retention get +--- + +## klio retention get + +Gets the currently applied retention policy + +``` +klio retention get [flags] +``` + +### Options + +``` + -h, --help help for get +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio retention](klio_retention.md) - Manage the retention policy + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_retention_set.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_retention_set.md new file mode 100644 index 00000000..8aa84a24 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_retention_set.md @@ -0,0 +1,45 @@ +--- +title: klio retention set +--- + +## klio retention set + +Sets the currently applied retention policy + +``` +klio retention set [flags] +``` + +### Options + +``` + -h, --help help for set + --keep-annual int Number of most recent annual backup kept + --keep-daily int Number of most recent daily backup kept + --keep-hourly int Number of most recent hourly backup kept + --keep-latest int Number of most recent latest backup kept + --keep-monthly int Number of most recent monthly backup kept + --keep-weekly int Number of most recent weekly backup kept +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio retention](klio_retention.md) - Manage the retention policy + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_send-wal.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_send-wal.md new file mode 100644 index 00000000..ca964f67 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_send-wal.md @@ -0,0 +1,40 @@ +--- +title: klio send-wal +--- + +## klio send-wal + +Upload the cluster's WALs to the target Klio server + +``` +klio send-wal [flags] +``` + +### Options + +``` + -h, --help help for send-wal + --primary Wait for the current instance to become a primary (default true) +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio](klio.md) - PostgreSQL Backup & Recovery for CloudNativePG + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_server.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_server.md new file mode 100644 index 00000000..3e7d9aea --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_server.md @@ -0,0 +1,36 @@ +--- +title: klio server +--- + +## klio server + +Starts and manage a Klio server + +### Options + +``` + -h, --help help for server +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio](klio.md) - PostgreSQL Backup & Recovery for CloudNativePG +* [klio server start](klio_server_start.md) - Starts a Klio server + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_server_start.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_server_start.md new file mode 100644 index 00000000..e371942d --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_server_start.md @@ -0,0 +1,42 @@ +--- +title: klio server start +--- + +## klio server start + +Starts a Klio server + +``` +klio server start [flags] +``` + +### Options + +``` + -h, --help help for start + --socket-path string Unix socket used by the administration server (default "/tmp/.klio-admin") + --tier1 Enables Tier1 server components (default true) + --tier2 Enables Tier2 server components +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio server](klio_server.md) - Starts and manage a Klio server + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_wal-player.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_wal-player.md new file mode 100644 index 00000000..0462e9d5 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_wal-player.md @@ -0,0 +1,37 @@ +--- +title: klio wal-player +--- + +## klio wal-player + +WAL Player Commands + +### Options + +``` + -h, --help help for wal-player +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio](klio.md) - PostgreSQL Backup & Recovery for CloudNativePG +* [klio wal-player generate](klio_wal-player_generate.md) - Generate a directory of WAL files +* [klio wal-player play](klio_wal-player_play.md) - Send to Klio a directory of WAL files + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_wal-player_generate.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_wal-player_generate.md new file mode 100644 index 00000000..dee705c2 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_wal-player_generate.md @@ -0,0 +1,41 @@ +--- +title: klio wal-player generate +--- + +## klio wal-player generate + +Generate a directory of WAL files + +``` +klio wal-player generate [output-directory] [flags] +``` + +### Options + +``` + -h, --help help for generate + --length int How many WAL files should be generated. Required. + --wal-size int The WAL file size in MBs. Defaults to 16 MB (default 16) +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio wal-player](klio_wal-player.md) - WAL Player Commands + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_wal-player_play.md b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_wal-player_play.md new file mode 100644 index 00000000..c48fb5b4 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/cli/klio_wal-player_play.md @@ -0,0 +1,41 @@ +--- +title: klio wal-player play +--- + +## klio wal-player play + +Send to Klio a directory of WAL files + +``` +klio wal-player play [directory] [flags] +``` + +### Options + +``` + --block-size int Block size in KiB. Max 8192 (8 MiB). Defaults to 2048. (default 2048) + -h, --help help for play + -j, --jobs int Number of parallel jobs to use when sending WALs (default 1) +``` + +### Options inherited from parent commands + +``` + --config string config file (default is $HOME/.klio.yaml) + --debug enable debug logging + --log-destination string where the log stream will be written + --log-field-level string JSON log field to report severity in (default: level) + --log-field-timestamp string JSON log field to report timestamp in (default: ts) + --log-level string the desired log level, one of error, info, debug and trace (default "info") + --pprof-server string enable the PPROF server using the specified address + --zap-devel Development Mode defaults(encoder=consoleEncoder,logLevel=Debug,stackTraceLevel=Warn). Production Mode defaults(encoder=jsonEncoder,logLevel=Info,stackTraceLevel=Error) + --zap-encoder encoder Zap log encoding (one of 'json' or 'console') + --zap-log-level level Zap Level to configure the verbosity of logging. Can be one of 'debug', 'info', 'error', 'panic' or any integer value > 0 which corresponds to custom debug levels of increasing verbosity + --zap-stacktrace-level level Zap Level at and above which stacktraces are captured (one of 'info', 'error', 'panic'). + --zap-time-encoding time-encoding Zap time encoding (one of 'epoch', 'millis', 'nano', 'iso8601', 'rfc3339' or 'rfc3339nano'). Defaults to 'epoch'. +``` + +### SEE ALSO + +* [klio wal-player](klio_wal-player.md) - WAL Player Commands + diff --git a/documentation/web/versioned_docs/version-0.0.18/user/grafana-dashboards.md b/documentation/web/versioned_docs/version-0.0.18/user/grafana-dashboards.md new file mode 100644 index 00000000..4a02de38 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/grafana-dashboards.md @@ -0,0 +1,174 @@ +--- +sidebar_position: 9 +--- + +# Grafana Dashboards + +Klio ships a Grafana dashboard template that visualizes the metrics it +exposes. The dashboard is generated from code with the +[Grafana Foundation SDK](https://github.com/grafana/grafana-foundation-sdk) +and committed to the repository at +`observability/grafana/klio-dashboard.json`. + +The dashboard queries the Prometheus export of Klio's OpenTelemetry metrics, +so it complements the [OpenTelemetry](opentelemetry.md) setup rather than +replacing it. + +## What the dashboard shows + +The dashboard is a single dashboard split into row sections: + +- **Client / Plugin** — the backup lifecycle as seen by the plugin sidecar + running in each PostgreSQL pod: backups in progress, time since the last + backup started, succeeded and failed, the latest backup duration, the + p50/p95/p99 backup duration distribution, backup run and failure rates, + and the backup success ratio. Also the WAL + streaming client the sidecar supervises as a child process: the PostgreSQL + timeline it is currently streaming and the p50/p95/p99 latency of sending + a WAL block to the server. + +![Klio client and plugin metrics](images/klio_client_and_plugin_metrics.png) + +- **Server** — the state of the Klio server StatefulSet: uptime, WAL ingest + throughput and freshness per tier, the latest written LSN, backup + verification outcomes, the base snapshot inventory (including file and + directory counts), p50/p95/p99 WAL block/get/upload duration by path, + stage and tier, tier-2 relay and maintenance run rates, the retention + window of the + physical PostgreSQL backups (counts by tier, latest/oldest backup age, + start/end LSN and PostgreSQL timeline per cluster), and the embedded NATS + JetStream queue. + +![Klio server metrics](images/klio_server_metrics.png) + +- **WAL Replication Lag** — how far Klio's WAL streaming client trails the + PostgreSQL primary, using CloudNativePG's replication metrics: the replay + lag in bytes and the flush lag in seconds. These panels read the + `cnpg_pg_stat_replication_*` metrics, so they require CloudNativePG + monitoring to be scraped into the same Prometheus (see the prerequisites + below). + +![Klio WAL replication lag metrics](images/klio_wal_replication_lag_metrics.png) + +Some panels need extra context to interpret correctly. Two are derived +from the alerting guidance in [OpenTelemetry](opentelemetry.md): + +- **Time since last WAL written by tier** surfaces the staleness signal + described under *Alerting on stalled WAL processing*: a stale tier-1 value + means PostgreSQL is no longer shipping WALs, while a stale tier-2 value + means the remote backend is no longer receiving them. +- **Tier-2 archival backlog (LSN gap)** plots the LSN difference between + tier 1 (local disk) and tier 2 (remote storage). Read together with the + staleness panel, it tells a slow pipeline (timestamps advancing, gap + growing) apart from a stalled one (timestamps and LSN both frozen). + +Two more are a statistical caveat rather than an alerting signal. Both are +histogram percentiles that need enough recent samples to be reliable: + +- **WAL block send duration (p50/p95/p99) by cluster** is most meaningful + under active write load. On an idle or low-write cluster, WAL blocks are + sent too infrequently for the underlying `histogram_quantile` to produce + a reliable percentile, so the line can look sparse or noisy rather than + simply absent. +- **Backup duration (p50/p95/p99)** has the same limitation, more acutely: + backups are infrequent, so this panel is computed over the whole selected + range (rather than a short rate window) to stay populated between runs. + Widen the dashboard range to span several backups for a stable reading; if + the selected range contains no backup, the panel is empty. Use it to spot + backup runtime trending up over time rather than to read an instantaneous + value. + +## Prerequisites + +The dashboard reads the Prometheus names of Klio's metrics (for example +`klio_plugin_backup_runs_total` and `klio_server_wal_written_total`). You +therefore need Prometheus scraping those metrics. Any of the export paths +described in [OpenTelemetry](opentelemetry.md) works: + +- An OpenTelemetry Collector with a Prometheus exporter that Prometheus + scrapes. +- The Klio Prometheus exporter (`OTEL_METRICS_EXPORTER=prometheus`), + scraped directly. + +The **WAL Replication Lag** row additionally reads CloudNativePG's +`cnpg_pg_stat_replication_*` metrics. To populate it, scrape the +CloudNativePG cluster monitoring (its `PodMonitor`) into the same +Prometheus. The rest of the dashboard works without it. + +:::note +When you route metrics through an OpenTelemetry Collector, enable +`resource_to_telemetry_conversion` on the Prometheus exporter so that +resource attributes such as the pod and namespace become Prometheus labels. +The sample collector under +`operator/config/samples/opentelemetry/otel_collector.yaml` already does +this. +::: + +## Importing the dashboard + +1. In Grafana, go to **Dashboards → New → Import**. +1. Upload `observability/grafana/klio-dashboard.json` or paste its contents. +1. When prompted, select your Prometheus data source for the `datasource` + variable. + +The dashboard declares a `datasource` template variable, so it is portable +across Grafana installations and is not tied to a specific data source UID. +The `namespace` and `cluster` template variables at the top filter the panels +by Kubernetes namespace and PostgreSQL cluster. + +### Example: kube-prometheus-stack + +This example follows the same flow as the +[CloudNativePG quickstart](https://cloudnative-pg.io/docs/current/quickstart/#grafana-dashboard), +using the +[kube-prometheus-stack](https://github.com/prometheus-community/helm-charts/tree/main/charts/kube-prometheus-stack) +chart. Adapt it to your own Prometheus/Grafana if you run a different setup — +only the metric prerequisites described above are required. + +Install Prometheus and Grafana: + +```sh +helm repo add prometheus-community \ + https://prometheus-community.github.io/helm-charts +helm upgrade --install \ + -f https://raw.githubusercontent.com/cloudnative-pg/cloudnative-pg/main/docs/src/samples/monitoring/kube-stack-config.yaml \ + prometheus-community prometheus-community/kube-prometheus-stack +``` + +Ensure Prometheus scrapes Klio's metrics by deploying a `ServiceMonitor` (or a +`PodMonitor`, if the collector's `Service` has no labels) for the +OpenTelemetry collector's Prometheus exporter — see +`operator/config/samples/opentelemetry/otel_collector_svc_monitor.yaml`. + +Port-forward Grafana and log in with `admin` / `prom-operator`: + +```sh +kubectl port-forward svc/prometheus-community-grafana 3000:80 +``` + +Open `http://localhost:3000/` and import +`observability/grafana/klio-dashboard.json` via **Dashboards → New → Import**, +selecting your Prometheus data source. + +Alternatively, load it automatically through the Grafana dashboard sidecar +with a labeled `ConfigMap`: + +```sh +kubectl create configmap klio-grafana-dashboard \ + --from-file=klio-dashboard.json=observability/grafana/klio-dashboard.json +kubectl label configmap klio-grafana-dashboard grafana_dashboard=1 +``` + +## Regenerating the dashboard + +The committed JSON is generated from the Go program under +`observability/grafana/`. To regenerate it after changing the generator (or +after a `grafana-foundation-sdk` bump), run from the repository root: + +```sh +task grafana:gen +``` + +CI runs `task grafana:uncommitted`, which regenerates the dashboard and +fails if the committed JSON has drifted from the generator output. Commit +the regenerated file whenever it changes. diff --git a/documentation/web/versioned_docs/version-0.0.18/user/helm_chart.mdx b/documentation/web/versioned_docs/version-0.0.18/user/helm_chart.mdx new file mode 100644 index 00000000..a6e758c6 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/helm_chart.mdx @@ -0,0 +1,306 @@ +--- +title: Klio Operator Helm Chart +sidebar_position: 90 +--- + + +import PartialValues from './_helm_chart_values.md'; + + + +The Klio Operator Helm chart allows you to deploy the Klio +Operator in your Kubernetes cluster. It is distributed as a private +OCI image. + +:::info Namespace Selection +We have used the `cnpg-system` namespace for all examples on this page. + +If you have deployed CloudNativePG in a different namespace, ensure you replace +`cnpg-system` with your specific namespace in all commands. +The Klio Operator must reside in the same namespace as the CloudNativePG +operator. +::: + +## Prerequisites + +Before installing the Klio Operator, ensure you have: + +- **Helm** – see the [Helm installation guide](https://helm.sh/docs/intro/install/) +- **Kubernetes** cluster with appropriate permissions +- **CloudNativePG Operator** already installed in your Kubernetes cluster. + See the [CloudNativePG installation guide](https://cloudnative-pg.io/documentation/current/installation_upgrade/). +- **cert-manager** (optional, but strongly recommended for managing TLS certificates). + See the [cert-manager installation guide](https://cert-manager.io/docs/installation/). +- **Prometheus Operator** (optional, for operator monitoring). + See the [Prometheus Operator installation guide](https://prometheus-operator.dev/docs/getting-started/installation/). + +## Installation + +### Step 1: Create a values file + +Create a `values.yaml` file with your configuration, if you need to +override some default values. If you have not +installed cert-manager or the Prometheus Operator, disable them to +avoid errors during installation: + +```yaml +# Uncomment if cert-manager is not installed +# certmanager: +# enable: false + +# Uncomment if the Prometheus Operator is not installed +# prometheus: +# enable: false +``` + +See the [Configuration](#configuration) section for the full list of +available parameters. You will reuse this file for future upgrades. + +### Step 2: Install the Helm Chart + +Deploy the Klio Operator to your cluster: + + +```sh +helm install klio-operator \ + oci://ghcr.io/cloudnative-pg/klio-operator-chart \ + --version 0.0.18 \ + --namespace cnpg-system \ + -f values.yaml +``` + + +### Step 3: Verify Installation + +After installation, verify that the Klio Operator is running: + +```sh +kubectl get pods -n cnpg-system -l app.kubernetes.io/name=klio +``` + +You should see the operator pod in a `Running` state. Check the logs to ensure +there are no errors: + +```sh +kubectl logs -n cnpg-system -l app.kubernetes.io/name=klio -f +``` + +Verify that the Custom Resource Definitions (CRDs) were created: + +```sh +kubectl get crds | grep klio.cnpg.io +``` + +You should see CRDs like `servers.klio.cnpg.io` and +`pluginconfigurations.klio.cnpg.io`. + +## Configuration + +### Inspecting the Chart + +You can download the Helm chart to inspect its contents, review the +default values, and understand what resources it will create: + + +```sh +helm pull oci://ghcr.io/cloudnative-pg/klio-operator-chart \ + --version 0.0.18 \ + --untar +``` + + +This downloads the chart and extracts it into the +`klio-operator-chart` folder, where you can review the templates, +the default `values.yaml`, and other chart files. + +### Configuration Reference + + + +### CNPG API Group Detection + +Klio works with both [CloudNativePG](https://cloudnative-pg.io/) (API group +`postgresql.cnpg.io`) and CNPG-based operators that use a different API group. +The operator detects the CNPG API group and version automatically from each +`Cluster` resource at runtime — no configuration is required. + +## Upgrades + +A complete upgrade consists of three steps that must be performed in +order: + +1. [Upgrade the CRDs](#upgrade-the-crds) — apply updated CRD manifests +2. [Upgrade the Operator](#upgrade-the-operator) — run `helm upgrade` +3. [Upgrade the Klio Server and client](#upgrade-the-klio-server-and-client) — upgrade operand images + +### Upgrade the CRDs + +:::warning Upgrading CRDs +Helm does not upgrade CRDs after initial installation +([details](https://helm.sh/docs/topics/charts/#custom-resource-definitions-crds)). +When upgrading to a version that includes CRD changes, apply them +manually before running `helm upgrade` (as explained below). +::: + +:::warning CRD breaking changes +If the CRDs contain breaking changes, old resource definitions may +not work. In this case, resources should be edited when +[upgrading the Klio Server and client](#upgrade-the-klio-server-and-client). +Check the [Upgrade Notes](upgrade_notes.md) for version-specific +instructions. +::: + +#### Step 1: Download the new chart + +Download the new version of the Helm chart to access the updated CRD +manifests: + + +```sh +helm pull oci://ghcr.io/cloudnative-pg/klio-operator-chart \ + --version 0.0.18 \ + --untar +``` + + +#### Step 2: Review CRD changes + +Before applying, review the CRD changes to understand what is being +modified: + +```sh +kubectl diff -f klio-operator-chart/crds/ +``` + +#### Step 3: Apply the updated CRDs + +Apply the CRD manifests to your cluster: + +```sh +kubectl apply --server-side --force-conflicts -f klio-operator-chart/crds/ +``` + +#### Step 4: Clean up + +Remove the extracted chart directory: + +```sh +rm -rf klio-operator-chart +``` + +### Upgrade the Operator + +To upgrade the Klio Operator to a newer version, run the following +Helm upgrade command: + + +```sh +helm upgrade klio-operator \ + oci://ghcr.io/cloudnative-pg/klio-operator-chart \ + --version 0.0.18 \ + --namespace cnpg-system \ + -f values.yaml +``` + + +See the +[Helm upgrade documentation](https://helm.sh/docs/helm/helm_upgrade/#options) +for more details on this and other upgrade options. + +### Upgrade the Klio Server and client + +After upgrading the operator and the CRDs, Klio images in `Server` +resources and the ones used as sidecars in `Cluster` resources should +be upgraded as well. We recommend using the same version for the +operator and the operand. + +:::info Future improvement +We plan to reduce the amount of manual work required during upgrades in +future versions of the operator. +::: + +#### Upgrade Klio servers + +Klio servers are not automatically updated when a new operand image is +released. Update the `spec.image` field with the new image reference: + + +```sh +kubectl patch server -n \ + --type merge \ + -p '{"spec":{"image":"ghcr.io/cloudnative-pg/klio:v0.0.18"}}' +``` + + +Verify that the Klio Server pods are running the latest version: + +```sh +kubectl get pods -l klio.cnpg.io/klio-server -A \ + -o custom-columns='NAMESPACE:.metadata.namespace,NAME:.metadata.name,IMAGE:.spec.containers[0].image' +``` + +:::warning +If the Klio Pod gets into a broken state before the image upgrade, a new Pod +with the image upgrade +[will not be rolled out](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#forced-rollback). +In this case you should delete the Klio Pod. The StatefulSet managing it +will recreate it with the chosen image. +::: + +:::warning +Not all parameters can be updated in a StatefulSet. In case a new version of +the operator requires such a change, delete the StatefulSet managing the Klio +Server. The operator will automatically recreate the StatefulSet. +StatefulSets [retain their PVCs by default](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#persistentvolumeclaim-retention), +make sure your data is safe before proceeding. + +```sh +kubectl delete statefulset -l klio.cnpg.io/klio-server= -n +``` +::: + +#### Upgrade Klio client sidecars + +The Pods of the CNPG instances need to be recreated to run the new Klio image. +You can use the CNPG `kubectl` plugin to trigger a rolling update: + +```sh +kubectl cnpg restart +``` + +Refer to the +[CloudNativePG documentation](https://cloudnative-pg.io/documentation/current/rolling_update/) +for guidance on performing rolling updates. + +Verify that the sidecar containers are running the latest version: + +```sh +kubectl get pods -l cnpg.io/cluster -A \ + -o jsonpath='{range .items[*]}{.metadata.namespace}{"\t"}{.metadata.name}{"\t"}{range .spec.initContainers[?(@.name=="klio-plugin")]}{.image}{end}{"\n"}{end}' +``` + +## Uninstalling + +To uninstall the Klio Operator: + +```sh +helm uninstall klio-operator --namespace cnpg-system +``` + +:::warning Data Preservation +Uninstalling the operator does not automatically remove: + +- Custom Resource Definitions (CRDs) +- Existing Klio resources (Servers, PluginConfigurations) +- Persistent volumes containing backup data + +To completely remove Klio from your cluster, you must manually delete these +resources. +::: + +To remove the CRDs after uninstalling: + +```sh +kubectl delete crd servers.klio.cnpg.io +kubectl delete crd pluginconfigurations.klio.cnpg.io +``` diff --git a/documentation/web/versioned_docs/version-0.0.18/user/images/basebackups_walarchive.png b/documentation/web/versioned_docs/version-0.0.18/user/images/basebackups_walarchive.png new file mode 100644 index 00000000..868cef2e Binary files /dev/null and b/documentation/web/versioned_docs/version-0.0.18/user/images/basebackups_walarchive.png differ diff --git a/documentation/web/versioned_docs/version-0.0.18/user/images/klio_client_and_plugin_metrics.png b/documentation/web/versioned_docs/version-0.0.18/user/images/klio_client_and_plugin_metrics.png new file mode 100644 index 00000000..038e2a3c Binary files /dev/null and b/documentation/web/versioned_docs/version-0.0.18/user/images/klio_client_and_plugin_metrics.png differ diff --git a/documentation/web/versioned_docs/version-0.0.18/user/images/klio_server_metrics.png b/documentation/web/versioned_docs/version-0.0.18/user/images/klio_server_metrics.png new file mode 100644 index 00000000..0357dbad Binary files /dev/null and b/documentation/web/versioned_docs/version-0.0.18/user/images/klio_server_metrics.png differ diff --git a/documentation/web/versioned_docs/version-0.0.18/user/images/klio_wal_replication_lag_metrics.png b/documentation/web/versioned_docs/version-0.0.18/user/images/klio_wal_replication_lag_metrics.png new file mode 100644 index 00000000..7cd48860 Binary files /dev/null and b/documentation/web/versioned_docs/version-0.0.18/user/images/klio_wal_replication_lag_metrics.png differ diff --git a/documentation/web/versioned_docs/version-0.0.18/user/images/overview-multi-tiers.png b/documentation/web/versioned_docs/version-0.0.18/user/images/overview-multi-tiers.png new file mode 100644 index 00000000..feca5a85 Binary files /dev/null and b/documentation/web/versioned_docs/version-0.0.18/user/images/overview-multi-tiers.png differ diff --git a/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-namespace-multi.png b/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-namespace-multi.png new file mode 100644 index 00000000..77fe47b7 Binary files /dev/null and b/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-namespace-multi.png differ diff --git a/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-namespace-single.png b/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-namespace-single.png new file mode 100644 index 00000000..5a7f2f3e Binary files /dev/null and b/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-namespace-single.png differ diff --git a/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-shared-multi.png b/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-shared-multi.png new file mode 100644 index 00000000..4a4f5405 Binary files /dev/null and b/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-shared-multi.png differ diff --git a/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-shared-single.png b/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-shared-single.png new file mode 100644 index 00000000..d0adf2e4 Binary files /dev/null and b/documentation/web/versioned_docs/version-0.0.18/user/images/tier1-shared-single.png differ diff --git a/documentation/web/versioned_docs/version-0.0.18/user/images/wal-streaming.png b/documentation/web/versioned_docs/version-0.0.18/user/images/wal-streaming.png new file mode 100644 index 00000000..6a78ca99 Binary files /dev/null and b/documentation/web/versioned_docs/version-0.0.18/user/images/wal-streaming.png differ diff --git a/documentation/web/versioned_docs/version-0.0.18/user/index.mdx b/documentation/web/versioned_docs/version-0.0.18/user/index.mdx new file mode 100644 index 00000000..23f07e45 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/index.mdx @@ -0,0 +1,74 @@ +--- +sidebar_position: 1 +--- + +# Klio Overview + +:::warning +Klio is **experimental** and under active development. APIs, CRDs, and +behavior may change without notice, and it is not yet recommended for +production use. +::: + +**Klio** is a cloud-native solution for enterprise-grade backup and recovery of +PostgreSQL databases managed by [CloudNativePG](https://cloudnative-pg.io) on +Kubernetes. It is designed to handle: + +- The **Write-Ahead Log (WAL) archive** for a given PostgreSQL `Cluster` + resource, within the same Kubernetes namespace as the Klio deployment +- The **catalog of physical base backups** for that same cluster +- Optionally, multiple PostgreSQL clusters + +These critical backup artifacts are stored across two distinct storage tiers: + +- Tier 1 – **Local Volume**: A local Persistent Volume (PV) within the + same namespace as the associated `Cluster` resource. It offers immediate, + high-throughput access for backup and recovery operations. Also referred to as + the **Main Tier** or **Klio Server**. + +- Tier 2 – **Secondary Storage**: An external object storage system where data + from Tier 1 is asynchronously replicated. This tier typically resides outside + the Kubernetes cluster, enabling geographical redundancy and enhancing disaster + recovery (DR) resilience. + +![Multi-tiered architecture overview](images/overview-multi-tiers.png) + +--- + +## Key Features + +### WAL Management + +- Native WAL streaming from the primary, eliminating the need for + `archive_command`, with support for: + - Partial WAL file handling + - WAL file compression + - WAL file encryption using user-provided keys + - Controlled replication slot advancement to ensure uninterrupted streaming + - Synchronous replication +- WAL archive storage on a local PVC (Tier 1) +- Extension of base backup retention policy enforcement to WAL files +- Asynchronous WAL relay to Tier 2 object storage + +:::note +Klio's WAL management utilizes the `READ_REPLICATION_SLOT` streaming +replication command, which was introduced in PostgreSQL 15. +Therefore, Klio requires PostgreSQL version 15 or greater to function properly. +::: + +### Base Backup Catalog + +- Physical online base backups from the primary node to Tier 1, with support + for: + - Data deduplication for efficient remote incremental backups + - Encryption using user-provided keys for data confidentiality +- Backup catalog stored on a file system Persistent Volume Claim (PVC) in Tier 1 +- Retention policy enforcement +- Asynchronous replication of base backups to Tier 2 object storage for + long-term durability and disaster recovery + +### General Capabilities + +- End-to-end encryption: both in-transit and at-rest +- Delivered as a CNPG-I plugin, with an accompanying Kubernetes Operator +- Distributed via a Helm chart for streamlined deployment diff --git a/documentation/web/versioned_docs/version-0.0.18/user/klio_server.md b/documentation/web/versioned_docs/version-0.0.18/user/klio_server.md new file mode 100644 index 00000000..3c8830bd --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/klio_server.md @@ -0,0 +1,968 @@ +--- +sidebar_position: 5 +--- + +# The Klio Server + +The Klio server is a central component of the Klio backup solution. It is +defined as the `Server` custom resource in Kubernetes, which creates a +StatefulSet running the Klio server application. + +The Klio server runs as a single `server` container. On startup, it first +initializes the Kopia repository, then starts serving both base backups +(using Kopia) and the incoming stream of PostgreSQL Write-Ahead Logs (WAL). + +The base backups and WAL files are stored on a single PersistentVolume attached +to the Klio server pod, in the `/data/base` and `/data/wal` directories, +respectively. + +## Storage Tiers + +### Tier 1: Local Storage + +Tier 1 uses local `PersistentVolumes` for immediate data access. +This is the primary landing zone for backups and WAL files, +providing the fastest recovery times. + +### Tier 2: Remote Object Storage + +Tier 2 offloads data to object storage for long-term retention and disaster +recovery. When Tier 2 is enabled alongside Tier 1, the server uses a work +queue to manage the asynchronous transfer of data from local storage to object +storage. + +Alternatively, you can deploy a **read-only server** with only Tier 2 +configured. This is useful for disaster recovery sites that need to restore +from object storage without the overhead of local storage. See the +[Read-Only Mode](#read-only-mode) section for details. + +Currently, Klio supports only Amazon S3 and S3-compatible storage providers. +See the [Object Store](#object-store) section for configuration details. + +### The Work Queue + +When Tier 1 is configured, the Klio Server pods will use a work queue. +The work queue is backed by NATS JetStream with file storage on a separate +`PersistentVolume` mounted at `/queue`. +The queue serves two purposes: + +- **Retention policy enforcement**: Tracks which WAL files are in use before + deletion +- **Tier 2 replication**: When Tier 2 is enabled, manages asynchronous + transfer to object storage + +## Storage Requirements + +The Klio Server uses multiple PersistentVolumeClaims (PVCs), each +serving a different purpose. Understanding what each PVC contains helps you +size them appropriately for your environment. For guidance on managing +storage capacity and resizing PVCs, see +[Managing Storage](managing_storage.md). + +### Data PVC + +The data PVC stores all backup data and WAL archives for Tier 1 storage. + +It holds the base backups and the WAL archive of all the servers that are backed +up. + +The following factors should be considered when defining the PVC size: + +1. WAL file production rate +1. Base backup size +1. Retention policies + +### Cache PVCs + +The cache PVCs (one for Tier 1 and Tier 2 each) are used by Kopia for its +[caching operations](https://kopia.io/docs/advanced/caching/). +They are used to speed up snapshot operations. + +:::warning +Klio is currently limited to use the default cache size when creating a Kopia +repository, 5GB for content and 5GB for metadata. +The cache sizes are not hard limits, as the cache is swept periodically, +so users should have a space buffer to account for this additional space. +This limitation will be removed in a future version. +::: + +### Queue PVC + +The queue PVC is required when Tier 1 is configured. It stores the NATS +JetStream work queue used for retention policy enforcement and asynchronous +Tier 2 replication. + +#### Queue Sizing Guidelines + +The queue stores only task metadata (cluster name and WAL filename), not the +actual WAL content. This means queue size depends on the **number of WAL +segments** generated, not the size of your database. + +**Sizing formula:** + +``` +Queue Size = WAL_segments_per_hour × max_backlog_hours × 300 bytes × 2 +``` + +Where: + +- **WAL_segments_per_hour**: How many WAL segments your database generates per + hour (check with `pg_stat_archiver` or monitor WAL production) +- **max_backlog_hours**: Maximum duration the Tier 2 WAL replication backlog + can grow before the queue fills up and tasks are lost. Backlog builds up + when Tier 2 replication falls behind Tier 1 ingestion — for example, during + Tier 2 outages or when object storage uploads are slower than local disk writes. +- **300 bytes**: Approximate storage per WAL task (message + JetStream overhead) +- **2**: Safety factor + +**Recommended sizes:** + +| Workload | WAL Rate | Recommended Size | +|----------|----------|------------------| +| Low write (OLTP) | ~60 segments/hour | **10 MiB** | +| Medium write | ~120 segments/hour | **25 MiB** | +| High write | ~360 segments/hour | **50 MiB** | +| Very high write | >500 segments/hour | **100 MiB** | + +These recommendations assume a 24-hour backlog tolerance and include an +additional **~10x safety margin** beyond the formula result. This margin accounts +for: + +- **Burst workloads**: WAL production can spike significantly above average rates +- **Multiple clusters**: A single Klio server may handle several CNPG clusters +- **Low cost of headroom**: Storage is cheap relative to the risk of queue + overflow, which causes WAL loss + +For shorter tolerance windows, you can reduce the queue size proportionally, but +keep the safety margin. + +:::tip +Start with 50 MiB as a conservative default. Monitor queue usage with the +`klio admin queue status` command and adjust based on actual WAL production +rates in your environment. +::: + +:::note +Large transactions that modify significant amounts of data will automatically +generate multiple WAL segments (PostgreSQL rotates WAL files at ~16 MB by +default). Account for this when estimating your WAL segment rate. +::: + +## Setting up a new Klio server + +Setting up a Klio server involves creating a `Server` resource along with the +required Kubernetes secrets and certificates. + +### Prerequisites + +Before setting up a Klio server, ensure you have: + +- A Kubernetes cluster with the Klio operator installed +- `kubectl` configured to access your cluster +- [cert-manager](https://cert-manager.io/) installed for certificate + management (recommended) +- [Age](https://github.com/FiloSottile/age) CLI installed locally + (for encrypting the backup encryption key) +- Enough storage resources for the data and cache PersistentVolumeClaims +- Enough storage resources for the queue PersistentVolumeClaim + +### Required Components + +A Klio server setup requires the following components: + +1. **Server Resource**: The main `Server` custom resource +1. **TLS Certificate**: For secure communication +1. **Encryption Key**: Age-encrypted key for backup data at rest +1. **CA Certificate**: For client authentication via mTLS +1. **Storage**: PersistentVolumeClaims for data, cache, and queue + +### Step-by-step setup + +#### 1. Create the Encryption Key + +The encryption key is used to encrypt backup data at rest. +Klio uses [Age](https://github.com/FiloSottile/age) encryption +to protect the key, enabling credential rotation without +touching the Kopia repository. + +See the [Age Encryption](#age-encryption) section for full +setup details. In summary: + +1. Generate an Age key pair (`age-keygen -o identity.txt`) +2. Generate and encrypt a random key + (`openssl rand -hex 32 | age -r -o key.age`) +3. Create Kubernetes Secrets for both files +4. Reference them in the Server spec + +:::tip +Use a strong, randomly generated key. This key is critical +for data security and recovery. +::: + +#### 2. Create CA Certificate + +Using cert-manager, a CA certificate can be created by using the following +Certificate resource: + +```yaml +--- +apiVersion: cert-manager.io/v1 +kind: Issuer +metadata: + name: selfsigned-issuer + namespace: default +spec: + selfSigned: { } +--- +apiVersion: cert-manager.io/v1 +kind: Certificate +metadata: + name: server-sample-ca +spec: + commonName: server-sample-ca + secretName: server-sample-ca + + duration: 2160h # 90d + renewBefore: 360h # 15d + + isCA: true + usages: + - cert sign + + issuerRef: + name: selfsigned-issuer + kind: Issuer + group: cert-manager.io +``` + +Apply the CA configuration with: + +```bash +kubectl apply -f ca-configuration.yaml +``` + +In the previous example, the CA to be used for authentication is signed by a +self-signed issuer. This doesn't pose any security issue as this CA is only +used internally and trust is established through configuration. + +The primary concern is the relationship between the client and the certificates +signed by the CA. + +:::info +The usage of a self-signed CA is not required by the Klio server. If your +PKI infrastructure already includes a CA for this scope, that CA can be used +for the Klio server, too. +::: + +#### 3. Create TLS Certificate + +Using cert-manager, create a self-signed certificate (for development) or use +your organization's certificate issuer: + +```yaml +--- +apiVersion: cert-manager.io/v1 +kind: Issuer +metadata: + name: selfsigned-issuer + namespace: default +spec: + selfSigned: { } +--- +apiVersion: cert-manager.io/v1 +kind: Certificate +metadata: + name: my-server-cert + namespace: default +spec: + secretName: my-server-tls + commonName: my-server + dnsNames: + - my-server + - my-server.default + - my-server.default.svc + - my-server.default.svc.cluster.local + duration: 2160h # 90 days + renewBefore: 360h # 15 days + isCA: false + usages: + - server auth + issuerRef: + name: selfsigned-issuer + kind: Issuer + group: cert-manager.io +``` + +Apply the certificate configuration: + +```bash +kubectl apply -f tls-certificate.yaml +``` + +:::info +For production environments, use certificates signed by your organization's +Certificate Authority (CA) or a trusted public CA instead of self-signed +certificates. +::: + +#### 4. Create the Server Resource + +Now create the main `Server` resource: + + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: Server +metadata: + name: my-server + namespace: default +spec: + # Container image for the Klio server + image: ghcr.io/cloudnative-pg/klio:v0.0.18 + imagePullPolicy: IfNotPresent + imagePullSecrets: [] # Add image pull secrets if needed + + # TLS configuration + tlsSecretName: my-server-tls + + # Client authentication configuration + caSecretName: server-sample-ca + + # Mode: standard (default) or read-only + # Omit this field or set to "standard" for normal read-write operation + # Set to "read-only" for DR/restore-only servers (see Read-Only Mode section) + # mode: standard + + # tier 1 configuration + tier1: + # Cache storage configuration + cache: + pvcTemplate: + storageClassName: standard # Adjust to your storage class (use 'kubectl get storageclass' to see available options) + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 10Gi # Adjust based on your needs + # Data storage pvcTemplate (for backups and WAL) + data: + pvcTemplate: + storageClassName: standard # Adjust to your storage class (use 'kubectl get storageclass' to see available options) + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 100Gi # Adjust based on your backup needs + # Age-encrypted encryption key file + encryptionKeyFile: + fileReference: + volume: + secret: + secretName: my-server-encryption-key + path: encryption-key.age + # Age identity file for decryption + identityFile: + fileReference: + volume: + secret: + secretName: my-server-age-identity + path: identity.txt + + # Queue storage configuration (for NATS work queue) + # Required when tier1 is configured + # See "Queue Sizing Guidelines" section for recommendations + queue: + pvcTemplate: + storageClassName: standard # Adjust to your storage class + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 50Mi # See Queue Sizing Guidelines; 50Mi suits most workloads + + # tier 2 configuration + tier2: + # Cache storage configuration + cache: + pvcTemplate: + resources: + requests: + storage: 1Gi + accessModes: + - ReadWriteOnce + # Age-encrypted encryption key file + encryptionKeyFile: + fileReference: + volume: + secret: + secretName: my-server-encryption-key + path: encryption-key.age + # Age identity file for decryption + identityFile: + fileReference: + volume: + secret: + secretName: my-server-age-identity + path: identity.txt + # S3 access configuration + s3: + prefix: klio + bucketName: klio-bucket + endpoint: https://rustfs:9000 + region: us-east-1 + accessKeyId: + name: rustfs + key: RUSTFS_ACCESS_KEY + secretAccessKey: + name: rustfs + key: RUSTFS_SECRET_KEY + customCaBundle: + name: rustfs-tls + key: tls.crt +``` + + +The example above uses credential-based S3 authentication. For alternative +authentication methods (IAM roles, IRSA, Pod Identity) and S3-compatible +storage providers, see the [Object Store](#object-store) section. + +Apply the Server resource: + +```bash +kubectl apply -f klio-server.yaml +``` + +#### 5. Verify the Server is Running + +Check the status of your Klio server: + +```bash +# Check the Server resource status +kubectl get server my-server -n default + +# Check the StatefulSet +kubectl get statefulset my-server-klio -n default + +# Check the Pod +kubectl get pods -l klio.cnpg.io/klio-server=my-server -n default + +# View logs +kubectl logs -l klio.cnpg.io/klio-server=my-server -n default -f +``` + +The server should create a StatefulSet with a pod named `my-server-klio-0`. + +## Read-Only Mode + +Klio servers can operate in read-only mode, allowing them to serve backups and +WAL files from Tier 2 object storage without accepting new backup writes. This +is useful for disaster recovery sites, cost-optimized restore-only deployments, +and multi-region architectures. + +### When to Use Read-Only Mode + +Use read-only mode when you need: + +- **Disaster recovery sites**: Deploy in secondary regions to restore from + shared S3 storage without duplicating backup writes +- **Geographic distribution**: Multiple read-only servers in different regions + can all restore from a single S3 bucket populated by one primary server +- **Read-only access control**: Prevent accidental backup modifications at + certain sites + +### Configuration + +A read-only server requires: + +- `mode: read-only` field in the spec +- `tier2` configuration (S3 object storage) +- **No** `tier1` configuration +- **No** `queue` configuration + +:::note +The `mode` field is immutable. Once a Server is created, its mode +cannot be changed. To operate in a different mode, you would need +another Klio server with a different mode. +::: + + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: Server +metadata: + name: dr-server + namespace: default +spec: + # Set mode to read-only + mode: read-only + + # Container image for the Klio server + image: ghcr.io/cloudnative-pg/klio:v0.0.18 + imagePullPolicy: IfNotPresent + + # TLS configuration + tlsSecretName: dr-server-tls + + # Client authentication configuration + caSecretName: server-sample-ca + + # Tier 2 configuration (required for read-only mode) + tier2: + # Cache storage configuration + cache: + pvcTemplate: + storageClassName: standard + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 10Gi # Only cache needed, no data storage + + # Age-encrypted encryption key file + encryptionKeyFile: + fileReference: + volume: + secret: + secretName: dr-server-encryption-key + path: encryption-key.age + # Age identity file for decryption + identityFile: + fileReference: + volume: + secret: + secretName: dr-server-age-identity + path: identity.txt + + # S3 access configuration + # See Object Store section for authentication options + s3: + bucketName: klio-backups + region: us-east-1 + accessKeyId: + name: s3-credentials + key: ACCESS_KEY_ID + secretAccessKey: + name: s3-credentials + key: SECRET_ACCESS_KEY +``` + + +Apply the read-only server: + +```bash +kubectl apply -f dr-server.yaml +``` + +### Using a Read-Only Server for Recovery + +Once deployed, PostgreSQL clusters can use the read-only server as a restore +source through a PluginConfiguration. The server will fetch backups and WAL +files from Tier 2 object storage transparently. + +See the Read-Only Server Mode section in the +[Architectures](architectures.md) documentation for detailed use cases and +architectural patterns. + +### Restrictions + +In read-only mode, the following operations are **not available**: + +- Creating new backups +- Sending WAL files to the server +- Applying retention policies +- Any write operations + +## Advanced Configuration + +The `.spec.template` field allows you to customize the Klio server's pod +template. You can add additional containers, volumes, or modify existing +settings. + +:::warning Advanced Users Only +The `.spec.template` field is primarily designed for advanced configurations. +While powerful, improper modifications can affect server functionality. +Always test changes in a non-production environment first. +::: + +:::note +The `containers` field within `.spec.template.spec` is mandatory but will be +merged with the default Klio `server` container. If you do not need to add +containers or modify the default one, you must still include an empty list. +::: + +### Node Affinity and Tolerations + +To dedicate specific nodes for Klio workloads (e.g., for performance isolation +or to separate backup workloads from application workloads), you can use the +`template` field to define affinity and toleration rules. + +```yaml +spec: + template: + spec: + # Mandatory field; merged with default containers + containers: [] + tolerations: + # Allow scheduling on nodes tainted for Klio + - key: node-role.kubernetes.io/klio + operator: Exists + effect: NoSchedule + affinity: + # Require nodes labeled for Klio + nodeAffinity: + requiredDuringSchedulingIgnoredDuringExecution: + nodeSelectorTerms: + - matchExpressions: + - key: node-role.kubernetes.io/klio + operator: Exists +``` + +See [Reserving Nodes for Klio Workloads](architectures.md#reserving-nodes-for-klio-workloads) +for details on node tainting. + +### Monitoring + +Refer to the [OpenTelemetry](./opentelemetry.md#klio-server-with-opentelemetry) +documentation for setting up monitoring and telemetry for the Klio server. + +## Object Store + +Klio uses object storage for Tier 2, providing durable, cost-effective +long-term backup storage. Currently, Klio supports Amazon S3 and S3-compatible +storage providers. + +### S3 + +Tier 2 is configured using the `tier2.s3` field in the Server spec. The +configuration is the same for both AWS S3 and S3-compatible providers. + +#### Basic Configuration with Credentials + +```yaml +tier2: + s3: + bucketName: klio-backups + region: us-east-1 + accessKeyId: + name: s3-credentials + key: ACCESS_KEY_ID + secretAccessKey: + name: s3-credentials + key: SECRET_ACCESS_KEY +``` + +#### S3-Compatible Storage with Custom Endpoint + +For S3-compatible providers, add the `endpoint` field: + +```yaml +tier2: + s3: + bucketName: klio-backups + endpoint: https://: + region: us-east-1 # May be required depending on provider + accessKeyId: + name: s3-credentials + key: ACCESS_KEY_ID + secretAccessKey: + name: s3-credentials + key: SECRET_ACCESS_KEY +``` + +#### Custom CA Certificates + +For providers using self-signed certificates or custom CAs: + +```yaml +tier2: + s3: + bucketName: klio-backups + endpoint: https://: + customCaBundle: + name: minio-ca-cert + key: ca.crt + accessKeyId: + name: s3-credentials + key: ACCESS_KEY_ID + secretAccessKey: + name: s3-credentials + key: SECRET_ACCESS_KEY +``` + +#### AWS IAM Roles (IRSA/Pod Identity) + +For AWS EKS clusters, using +[IAM Roles for Service Accounts (IRSA)](https://docs.aws.amazon.com/eks/latest/userguide/iam-roles-for-service-accounts.html) +or [EKS Pod Identity](https://docs.aws.amazon.com/eks/latest/userguide/pod-identities.html) +is the recommended approach. This provides better security through automatic +credential rotation, reduced secret sprawl, and fine-grained IAM policies. + +To use IAM role-based authentication: + +1. Create an IAM role with appropriate S3 permissions +1. Create a Kubernetes ServiceAccount with the IAM role annotation (for IRSA) + or Pod Identity association (for Pod Identity) +1. Reference the ServiceAccount in the Server spec and omit credentials: + +```yaml +spec: + tier2: + s3: + bucketName: klio-backups + region: us-east-1 + # No accessKeyId or secretAccessKey - use IAM role + + template: + spec: + serviceAccountName: klio-s3-access + containers: [] # Mandatory but merged with defaults +``` + +The AWS SDK will automatically use the pod's IAM role credentials when +`accessKeyId` and `secretAccessKey` are omitted. + +## Encryption + +Klio implements encryption at rest for both base backups and WAL files to +ensure data security throughout the backup lifecycle. + +### Base Backups Encryption + +Base backups are encrypted by Kopia using the encryption key +decrypted from the Age-encrypted key file. Kopia handles +encryption transparently. + +The encryption key is set during repository initialization and is required +for all subsequent backup and restore operations. + +:::warning Critical +Store the encryption key securely. Loss of this key means permanent +loss of access to all backup data. There is no key recovery mechanism. +::: + +### WAL Files Encryption + +WAL files are encrypted using a master key derivation system with authenticated +encryption. The encryption process works as follows: + +1. **Master Key Generation**: A 32-byte master key is derived from the encryption + key using PBKDF2 +1. **Key Enveloping**: The master key itself is encrypted using AES-256-GCM + with a password-derived encryption key to protect the key at rest +1. **Per-File Encryption**: Each WAL file is compressed and then encrypted using + the master key with authenticated encryption before being stored + +WAL files are first compressed using Snappy S2 compression, +then encrypted to ensure both space efficiency and security. + +The same encryption key used for base backups encrypts the WAL files, +ensuring a unified security model across all backup artifacts. + +### Encryption Credential Rotation + +The underlying encryption key (used by Kopia and the WAL keychain) +cannot be changed once set. However, you can rotate the Age identity +without touching the encryption key or the repository: + +1. Generate a new Age key pair +1. Re-encrypt the encryption key file with the new public key +1. Deploy the new identity file and re-encrypted key file + +This rotation only changes how the encryption key is protected, +not the key itself. See [Rotating Age Credentials](#rotating-age-credentials) +for step-by-step instructions. + +:::tip +Choose a strong encryption key from the start. Use a password +manager or key management system to generate and store a +cryptographically secure key (recommended: 32+ random characters). +::: + +### Encryption in Transit + +In addition to encryption at rest, Klio protects both base backups and WAL files +during transmission using TLS (Transport Layer Security). + +All communication between a Klio client and the Klio server is secured +with TLS: + +- **Base Backup Traffic**: Kopia client connections to the base backup server + are encrypted using TLS, protecting backup data as it transfers to the Klio + server +- **WAL Streaming**: PostgreSQL instances streaming WAL files to the Klio server + use gRPC over TLS, ensuring WAL data is encrypted during transmission + +The TLS certificate is configured via the `.spec.tlsSecretName` field in the +Server resource, which references a Kubernetes secret containing the TLS +certificate and private key. This provides end-to-end encryption, ensuring that +backup data is protected both at rest and in transit. + +### Age Encryption + +[Age](https://github.com/FiloSottile/age) is a modern file +encryption tool that Klio supports for protecting the encryption +key. Instead of storing the plaintext encryption key in a +Kubernetes Secret, you encrypt it with an Age public key and +provide the corresponding Age identity (private key) to Klio. + +This enables: + +- **Credential rotation** without touching the Kopia repository + or WAL data. +- **Multiple recipients** for disaster recovery or team access. +- **Offline operations** — re-encryption can be done with the + standard `age` CLI. + +#### Setup + +1. Generate an Age key pair: + +```bash +age-keygen -o identity.txt +# Public key: age1ql3z7hjy54pw3hyww5ayyfg7zqgvc7w3j2elw8zmrj2kg5sfn9aqmcac8p +``` + +2. Generate a random encryption key and encrypt it: + +```bash +openssl rand -hex 32 | age \ + -r age1ql3z7hjy54pw3hyww5ayyfg7zqgvc7w3j2elw8zmrj2kg5sfn9aqmcac8p \ + -o encryption-key.age +``` + +3. Create Kubernetes Secrets for both files: + +```bash +kubectl create secret generic klio-encryption-key-age \ + --from-file=encryption-key.age +kubectl create secret generic klio-age-identity \ + --from-file=identity.txt +``` + +4. Reference them in the Server spec: + +```yaml +tier1: + encryptionKeyFile: + fileReference: + volume: + secret: + secretName: klio-encryption-key-age + path: encryption-key.age + identityFile: + fileReference: + volume: + secret: + secretName: klio-age-identity + path: identity.txt +``` + +The same configuration applies to `tier2`. + +:::note +Only standard Age identities (X25519 keys) are supported. +Age plugins (e.g., `age-plugin-yubikey`) are not supported +directly, but you can encrypt the key file to both a +plugin-based recipient and a standard X25519 recipient. +::: + +#### Using External Secret Managers + +The `encryptionKeyFile` and `identityFile` fields accept any +Kubernetes `VolumeSource`, not just Secrets. This enables +integration with external secret management systems: + +```yaml +tier1: + encryptionKeyFile: + fileReference: + volume: + csi: + driver: secrets-store.csi.k8s.io + readOnly: true + volumeAttributes: + secretProviderClass: klio-aws-secrets + path: encryption-key.age +``` + +#### Rotating Age Credentials + +To rotate the Age identity without touching the encryption key +or the repository: + +1. Generate a new Age key pair: + +```bash +age-keygen -o new-identity.txt +``` + +2. Re-encrypt the key file with the new public key: + +```bash +age -d -i identity.txt encryption-key.age | \ + age -r -o encryption-key-new.age +``` + +3. Update the Kubernetes Secrets: + +```bash +kubectl create secret generic klio-encryption-key-age \ + --from-file=encryption-key.age=encryption-key-new.age \ + --dry-run=client -o yaml | kubectl apply -f - +kubectl create secret generic klio-age-identity \ + --from-file=identity.txt=new-identity.txt \ + --dry-run=client -o yaml | kubectl apply -f - +``` + +4. Restart the Klio server pod to pick up the new files. + +5. Securely delete the old identity and plaintext files. + +## Authentication + +Klio uses mTLS Authentication for securing access to both the base backup server +and the WAL streaming server. Authentication is handled by verifying the client +certificates against the CA certificate which has been created when configuring +the Klio server. + +### Creating a client-side certificate + +To create a client-side certificate, you need a issuer that will sign all the +certificates with a CA known by the Klio server. Supposing that such a issuer is +called `server-sample-ca` and available in the current namespace, you can create +a client certificate with the following Certificate object: + +```yaml +apiVersion: cert-manager.io/v1 +kind: Certificate +metadata: + name: client-sample-tls +spec: + secretName: client-sample-tls + commonName: klio@cluster-1 + + duration: 2160h # 90d + renewBefore: 360h # 15d + + isCA: false + usages: + - client auth + + issuerRef: + name: server-sample-ca + kind: Issuer + group: cert-manager.io +``` + +If used the example proposed in the [server configuration documentation +page](#2-create-ca-certificate), the issuer can be created with: + +```yaml +apiVersion: cert-manager.io/v1 +kind: Issuer +metadata: + name: server-sample-ca +spec: + ca: + secretName: server-sample-ca +``` diff --git a/documentation/web/versioned_docs/version-0.0.18/user/main_concepts.md b/documentation/web/versioned_docs/version-0.0.18/user/main_concepts.md new file mode 100644 index 00000000..e762771b --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/main_concepts.md @@ -0,0 +1,131 @@ +--- +sidebar_position: 2 +--- + +# Main Concepts + +Klio is built on top of two foundational technologies: + +* PostgreSQL's native physical backup infrastructure +* The CloudNativePG Interface (CNPG-I) for backup and recovery + +PostgreSQL has provided **native continuous backup and point-in-time recovery +(PITR) capabilities since version 8.0, released in 2005**, enabling reliable +disaster recovery and business continuity for mission-critical systems +worldwide. + +:::info + +PostgreSQL offers logical backups using tools like `pg_dump`, which generate a +logical representation of the database as SQL statements or data files. Logical +backups do not provide continuous protection or point-in-time recovery +capabilities. As a result, they are not suitable for **business continuity +scenarios** in mission-critical environments where minimizing downtime and data +loss is essential. + +::: + +At its core, [PostgreSQL’s continuous backup and recovery](https://www.postgresql.org/docs/current/continuous-archiving.html) +system uses **physical (file system level) copies** combined with **write-ahead +log (WAL) archiving**. +This approach enables consistent, recoverable backups while keeping systems +online, a strategy proven effective in production environments for over two +decades. + +In a PostgreSQL backup solution, the infrastructure typically consists of: + +- **WAL Archive**: A designated location for continuously archived WAL + (write-ahead log) files, preserving all changes made to the database to + support data durability and recovery. +- **Physical Base Backups**: A consistent copy of all data files used by + PostgreSQL (primarily the `PGDATA` directory and any tablespaces), forming + the foundational layer for any recovery operation. + +The diagram below illustrates the relationship between physical base backups +and the WAL archive over time: + +![Physical backups, WAL archive, and time](images/basebackups_walarchive.png) + + +--- + +## WAL Archive + +The WAL archive is central to **continuous backup** in PostgreSQL and is +essential for: + +- **Hot (Online) Backups**: Allowing physical base backups to be taken from any + node (primary or standby) without shutting down PostgreSQL, ensuring backups + can proceed without service disruption. +- **Point-in-Time Recovery (PITR)**: Enabling recovery to any precise moment + after the earliest available base backup, using archived WAL files to replay + transactions up to the desired recovery point. + +:::warning + +WAL archives on their own are insufficient for disaster recovery. +A **physical base backup is required** to restore a PostgreSQL cluster. + +::: + +Using a WAL archive significantly enhances the resilience of a PostgreSQL +system. WAL files can be fetched by any PostgreSQL instance for replication or +recovery, with archives typically retaining WAL segments longer than local +retention policies, ensuring historical data is preserved for PITR and disaster +recovery workflows. + +Klio receives WAL content from a PostgreSQL primary via streaming replication. + +--- + +## Physical base backups + +PostgreSQL supports **physical base backups** as the cornerstone of its +disaster recovery and PITR strategies. A base backup is a **consistent, file +system-level copy** of all data files used by a PostgreSQL cluster, including +the `PGDATA` directory and any additional tablespaces. + +Key properties of PostgreSQL base backups: + +- **Online (Hot) Backups**: Base backups can be taken while the database is + online, avoiding downtime. PostgreSQL maintains consistency during an online + backup by coordinating with its write-ahead logging system, ensuring a valid + restore point. +- **Foundation for PITR**: A base backup provides the starting point for + point-in-time recovery. After restoring the base backup, archived WAL files + are replayed to advance the system to a specific recovery target, allowing + precise restoration following accidental data loss or corruption. +- **Efficient Storage and Transport**: Base backups can be compressed and + streamed to external or object storage, supporting offsite and cloud-based + disaster recovery workflows. + +Klio leverages CNPG-I to coordinate the hot backup procedure, using +PostgreSQL’s `pg_backup_start` and `pg_backup_stop` concurrent API to ensure +consistency. It uses [Kopia](https://github.com/kopia/kopia/) to efficiently +transfer backup data across locations, ensuring backups are portable, +secure, and space-efficient. + +--- + +## Recovery + +In PostgreSQL, **recovery** is the process of restoring a database cluster from +a **physical base backup**, bringing it to a consistent state by replaying +**write-ahead log (WAL)** files, which contain the necessary *redo* information +for all changes made after the backup. + +PostgreSQL’s recovery system supports [Point-in-Time Recovery (PITR)](https://www.postgresql.org/docs/current/continuous-archiving.html#BACKUP-PITR-RECOVERY), +enabling you to restore a cluster to **any precise moment** between your +earliest base backup and the latest available WAL segment. To perform recovery, +a **valid WAL archive is required alongside the physical base backup**. + +Klio follows the approach of CloudNativePG and implements the recovery part of +CNPG-I. It **does not perform in-place recovery on an existing cluster**; +instead, recovery is used to **bootstrap a new cluster** from a base backup and +replay WAL files to reach a desired state. + +Recovery can operate in two primary modes: full recovery (replaying WAL files +to the latest available segment) or **Point-in-Time Recovery (PITR)**, allowing +restoration to a chosen state before an incident such as accidental data +deletion. Klio supports all PITR targets provided by CloudNativePG, including +time, restore point, and transaction. diff --git a/documentation/web/versioned_docs/version-0.0.18/user/managing_storage.md b/documentation/web/versioned_docs/version-0.0.18/user/managing_storage.md new file mode 100644 index 00000000..2fd5f971 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/managing_storage.md @@ -0,0 +1,291 @@ +--- +sidebar_position: 8 +--- + +# Managing Storage + +This guide explains how to manage storage on your Klio server, prevent +disk full scenarios, and recover when storage is exhausted. + +The Klio server uses persistent storage for backup data, WAL archives, cache, +and the work queue. When the data PVC approaches capacity, backup and WAL +archival operations may fail. + +## How Disk Space Is Freed + +Deleting a backup does **not** immediately free disk space. Klio is built +on top of [Kopia](https://kopia.io), which uses a two-phase approach: + +1. **Deletion** logically deletes backups +1. **Maintenance** actually removes deleted data, if the data is older + than 24 hours and unused by any existing backup + +This design prevents accidental data loss from concurrent operations. +However, it means that space is not freed until maintenance runs and the +24-hour safety window has passed. + +Maintenance runs automatically in the background. Klio does not currently +provide a way to trigger it on demand, but it can be started manually +by opening a shell on the Klio server pod and running: + +```bash +kopia maintenance set --owner=me \ + --config-file=/tmp/$KOPIACONFIG_TIER1_CONF \ + --disable-file-logging +kopia maintenance run \ + --config-file=/tmp/$KOPIACONFIG_TIER1_CONF \ + --full \ + --disable-file-logging +``` + +## When the Disk Is Full + +When the Klio data PVC is completely full: + +- All backup operations block (new backups, deletions, maintenance) +- WAL streaming to Klio stops +- PostgreSQL accumulates WAL files on its own PVC +- No backup corruption occurs, even when backups fail mid-operation +- No orphan or incomplete snapshots are left behind +- Existing backups remain intact and restorable + +All operations resume automatically when space is freed. + +:::warning +While the Klio disk is full, WAL files build up on the PostgreSQL PVC. +If this condition persists, it can lead to disk pressure on the database +side as well. Resolve Klio storage issues promptly to avoid cascading +failures. +::: + +## Resolving Storage Issues + +### Expand the PVC + +The simplest option is to expand the PVC. The Klio operator supports +expansion of PersistentVolumeClaims (PVCs) for all storage components: +data, cache (Tier 1 and Tier 2), and queue. + +#### Prerequisites + +PVC expansion requires a StorageClass with `allowVolumeExpansion: true`. + +:::warning User Responsibility +Before attempting to resize PVCs, **you must verify** that your +StorageClass supports volume expansion. The operator will attempt the +resize operation directly—if the StorageClass does not support +expansion, the Kubernetes API will reject the request and the operator +will log an error. +::: + +To check if your StorageClass supports expansion: + +```bash +kubectl get storageclass -o jsonpath='{.allowVolumeExpansion}' +``` + +If the output is not `true`, you need to either: +1. Update the StorageClass to enable volume expansion (if the underlying + storage provisioner supports it) +1. Use a different StorageClass that supports volume expansion +1. Migrate to a new PVC (see [Limitations](#limitations) for options) + +#### Expanding PVC Size + +To expand a PVC, update the corresponding +`pvcTemplate.resources.requests.storage` field in the Server spec with +a larger value: + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: Server +metadata: + name: my-server +spec: + tier1: + data: + pvcTemplate: + resources: + requests: + storage: 200Gi # Increased from 100Gi + cache: + pvcTemplate: + resources: + requests: + storage: 20Gi # Increased from 10Gi + queue: + pvcTemplate: + resources: + requests: + storage: 20Gi # Increased from 10Gi +``` + +Apply the updated Server resource: + +```bash +kubectl apply -f klio-server.yaml +``` + +#### What Happens During Resize + +When you update the Server spec with larger PVC sizes, the following +occurs: + +1. **PVC expansion**: The operator patches PVCs directly to the new + size. This modifies the PVC resources but does **not** update the + StatefulSet—the StatefulSet's VolumeClaimTemplates remain unchanged + at this point. +1. **Temporary misalignment**: After the PVC patch, there is a brief + period where the PVCs have the new size but the StatefulSet + VolumeClaimTemplates still reflect the old size. +1. **StatefulSet recreation**: The operator detects that the expected + StatefulSet (with new VolumeClaimTemplates) differs from the current + one. Since VolumeClaimTemplates are immutable in Kubernetes, the + StatefulSet is deleted and recreated to align with the new spec. +1. **Pod restart**: The Klio server pod restarts and mounts the + already-expanded PVCs. + +:::note Why explicit PVC patching is necessary +VolumeClaimTemplates only define specs for *new* PVCs—they do not resize +existing ones. Without explicit PVC patching by the operator, the +StatefulSet would be recreated but the PVCs would remain at their +original size, creating a permanent mismatch between the Server spec +and actual storage. +::: + +#### StatefulSet and PVC Alignment + +After the full resize operation completes: + +- The **PVCs** have the new expanded size +- The **StatefulSet VolumeClaimTemplates** match the new size (after + recreation) +- The **Server spec** is consistent with both + +This ensures no drift between the desired state and actual resources. + +#### Monitoring Resize Progress + +The operator emits a `PVCExpanded` Kubernetes event on the Server +resource when a PVC is successfully expanded. You can view these events +with: + +```bash +kubectl describe server my-server +``` + +Check the PVC status to monitor the resize operation: + +```bash +kubectl get pvc -l klio.cnpg.io/klio-server=my-server +``` + +The PVC will show the new requested size in +`spec.resources.requests.storage`. The actual capacity is reflected in +`status.capacity.storage` once the resize completes. + +For detailed status, including any resize conditions: + +```bash +kubectl describe pvc data-my-server-klio-0 +``` + +#### Limitations + +- **Expansion only**: PVC shrinking is not supported by Kubernetes. + Attempting to reduce the storage size will be ignored and logged as + a warning. +- **StorageClass support**: The StorageClass must have + `allowVolumeExpansion: true`. If the StorageClass does not support + expansion, the resize will fail and an error will be logged. +- **Pod restart required**: Due to StatefulSet VolumeClaimTemplates + being immutable, PVC expansion causes a brief pod restart. +- **Filesystem resize**: After the volume is expanded, the filesystem + must also be resized. Most modern storage providers handle this + automatically. + +:::warning No Automatic Fallback +If your StorageClass does not support volume expansion, there is **no +automatic fallback**. The operator will not delete and recreate PVCs to +achieve a larger size, as this would result in **permanent data loss**. +The only options in this case are: + +1. Migrate to a StorageClass that supports volume expansion +1. Create a new Klio server with larger PVCs and restore from backup +1. Manually migrate data (requires downtime and careful planning) + +::: + +### Delete Backups and Run Maintenance + +:::warning +Maintenance requires some free disk space to run. If the disk is +completely full, maintenance itself may fail. In that case, expand +the PVC first. +::: + +If PVC expansion is not available, free space by deleting old backups +and running maintenance manually. + +1. **Delete old backups:** + + ```bash + kubectl exec -it my-server-klio-0 -- \ + klio admin delete-backup \ + --cluster my-cluster --tier1 + ``` + + To delete from both tiers, add `--tier2`. + + :::note + The actual space freed depends on how much data is shared with + other backups through deduplication. Deleting a backup only + reclaims space for data blocks that are not referenced by any + remaining backup. + ::: + + :::warning + Deleting a backup removes the ability to restore to that point + in time. Ensure you have adequate backups remaining before + deletion. + ::: + +1. **Run maintenance** to reclaim space from deleted backups by opening + a shell on the Klio server pod and running: + + ```bash + kopia maintenance set --owner=me \ + --config-file=/tmp/$KOPIACONFIG_TIER1_CONF \ + --disable-file-logging + kopia maintenance run \ + --config-file=/tmp/$KOPIACONFIG_TIER1_CONF \ + --full \ + --disable-file-logging + ``` + +## Best Practices + +1. **Configure retention policies**: The most effective way to control + storage growth is through properly configured retention policies, which + automatically delete old backups and WAL files no longer needed for + recovery. See + [Retention Policies](plugin_configuration.md#retention-policies) for + configuration details. + +1. **Monitor storage usage**: Klio does not provide built-in storage + alerts. Set up monitoring and alerting on your PVC usage to detect + capacity issues before they cause failures. + +1. **Size Tier 1 storage appropriately**: Account for your backup + frequency, database size, change rate, and retention requirements when + provisioning the data PVC. Include buffer for the 24-hour window + during which deleted backup data is not yet eligible for garbage + collection. + +1. **Use Tier 2 for long-term retention**: Object storage (S3, etc.) is + more cost-effective and scales easily for long-term backup retention. + Keep Tier 1 lean for fast recovery of recent backups. + +1. **Use expandable StorageClasses**: When possible, use StorageClasses + with `allowVolumeExpansion: true` to enable online PVC expansion as a + recovery option. \ No newline at end of file diff --git a/documentation/web/versioned_docs/version-0.0.18/user/opentelemetry.md b/documentation/web/versioned_docs/version-0.0.18/user/opentelemetry.md new file mode 100644 index 00000000..5270968e --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/opentelemetry.md @@ -0,0 +1,746 @@ +--- +sidebar_position: 8 +--- + +# OpenTelemetry Observability + +Klio provides built-in support for [OpenTelemetry](https://opentelemetry.io/), +enabling comprehensive observability through distributed tracing and metrics +collection. This allows you to monitor backup operations, performance +characteristics, and system health across your Klio deployment. + +## Available Telemetry + +Klio automatically collects the following: + +- Traces + - Distributed WAL streaming and processing + - Backup lifecycle (backup, backup run, verification, maintenance) +- Metrics + - Server + - Server uptime + - Backup metrics + - Number of snapshots + - Number of files in the latest snapshot + - Number of directories in the latest snapshot + - Size of the latest snapshot + - Timestamp of the latest snapshot + - Timestamp of the oldest snapshot + - Number of retained PostgreSQL backups (per cluster and tier) + - Start/end time, timeline and LSN of the latest and oldest + retained PostgreSQL backup (per cluster and tier) + - Total number of backup verifications (split by outcome and tier) + - WAL processing metrics + - Number of WAL files written + - Bytes written + - Timestamp of the most recently written WAL file + - LSN progress of WAL ingestion (Tier 1) and archival (Tier 2) + - Timeline of the latest WAL on Tier 1 and Tier 2 + - Per-block processing durations split by stage (histogram) + - Per-file get and Tier 2 upload durations (histogram) + - Queue metrics + - Number of messages in the queue + - Number of bytes in the queue + - [GRPC metrics](https://opentelemetry.io/docs/specs/semconv/rpc/rpc-metrics/) + - [Go runtime statistics](https://pkg.go.dev/go.opentelemetry.io/contrib/instrumentation/runtime) + - [Host metrics](https://pkg.go.dev/go.opentelemetry.io/contrib/instrumentation/host) + - [Controller runtime metrics](https://book.kubebuilder.io/reference/metrics-reference) + - Sidecar + - Backup metrics + - Number of backups currently in progress + - Timestamp of the most recent backup start + - Timestamp of the most recent successful completion + - Timestamp of the most recent failure + - Duration of the most recent backup + - Total number of backup runs (split by outcome) + - Total number of backup verifications (split by outcome) + - [GRPC metrics](https://opentelemetry.io/docs/specs/semconv/rpc/rpc-metrics/) + - [Go runtime statistics](https://pkg.go.dev/go.opentelemetry.io/contrib/instrumentation/runtime) + - [Host metrics](https://pkg.go.dev/go.opentelemetry.io/contrib/instrumentation/host) + - [Controller runtime metrics](https://book.kubebuilder.io/reference/metrics-reference) + - Client (WAL streaming) + - Per-block WAL send durations (histogram) + - Timeline currently being streamed (gauge) + +:::note +Log exporters are not currently supported. +::: + +## Traces Reference + +### Backup lifecycle spans + +When a backup is triggered through CNPG-I, Klio creates +the following spans under the `klio.plugin.backup` tracer: + +| Span Name | Description | +|---|---| +| `backup` | Root span covering the entire backup operation (run + verify) | +| `backup_run` | Child span for the actual data backup execution | +| `backup_verify` | Child span for post-backup verification | + +The `backup` span includes the following attributes: + +| Attribute | Type | Description | +|---|---|---| +| `backup.name` | string | Name assigned to the backup | + +On failure, the span records the error and sets its status to +`ERROR`. + +### WAL streaming spans + +Klio traces WAL streaming at the per-file level; the per-block stage +timings that used to be spans are now recorded as the +[WAL duration histograms](#wal-duration-histograms-server) instead. + +| Span Name | Tracer | Description | +|---|---|---| +| `download_history_file` | `klio.client.wal` | Span for downloading a timeline history file (rare). | +| `get_wal` | `klio.server.wal` | Per-file span for the gRPC Get of a WAL file (served from tier-1 or tier-2). | +| `tier2_upload` | `klio.server.consumer` | One span per WAL file archived to tier-2 (remote storage). | + +### Kopia repository spans + +Klio runs Kopia as a subprocess to manage the deduplicated backup +repository. Kopia emits its own OpenTelemetry traces (snapshot uploads, +content and blob operations, and repository-server gRPC sessions) under +the `kopia` service name. Klio enables Kopia's trace exporter +automatically whenever its own traces are configured for the OTLP/gRPC +protocol, so that Kopia exports to the same collector as Klio. + +The individual span names come from the Kopia version bundled with Klio +and are Kopia internals rather than a Klio-defined contract; they may +change across Kopia upgrades. To see what your deployment emits, inspect +your tracing backend for traces whose `service.name` is `kopia` — for +example, the `OpenRepository` and `UploadDir` spans produced while a +backup runs. + +Unlike Klio's own telemetry, Kopia does not read `OTEL_SERVICE_NAME`, +`OTEL_RESOURCE_ATTRIBUTES`, or `OTEL_RESOURCE_DETECTORS`: every Kopia span +carries a fixed resource of just `service.name` (`kopia`) and +`service.version`. + +Kopia can only export traces over OTLP/gRPC. When traces are configured +for `http/protobuf`, Kopia tracing stays disabled and only Klio's own +spans are exported. When traces are configured for `grpc`, Klio automatically +enables Kopia tracing and exports both Klio and Kopia spans to the same +collector. You can override this automatic behavior with the +`KOPIA_ENABLE_OTLP_TRACE` variable in the same container environment: +set it to `false` to disable Kopia tracing even when Klio's own traces +use gRPC, or to `true` to force it on (Kopia still exports only over +gRPC, so the configured endpoint must be a gRPC one). + +Kopia traces are exported as independent traces, under the +`kopia` service, and are not correlated into Klio's trace tree. + +## Metrics Reference + +Klio metric names follow the +`klio...` taxonomy. The component +segment identifies which process emits the metric: + +- `plugin` — the CNPG plugin sidecar running in each PostgreSQL pod. +- `server` — the Klio server StatefulSet (hosts the Kopia server, the + WAL gRPC ingest, the embedded NATS JetStream queue, and the tier-2 + WAL consumer). +- `operator` — the Klio operator deployment. Bridges + controller-runtime Prometheus metrics to OTLP and adds Go + runtime and host instrumentation. +- `client` — the WAL streaming client that ships WAL from + PostgreSQL to the Klio server. Emits the per-block WAL send + duration histogram and the currently streamed timeline gauge. + +### Attributes + +Klio metrics carry the following attributes. Each per-metric table +below repeats the applicable attributes in its descriptions; this +section is the central reference for the value space of each +attribute key. + +| Attribute | Values | Applies to | +|---|---|---| +| `tier` | `tier1` (local disk on the Klio server), `tier2` (remote object store) | All `klio.server.wal.*` and `klio.server.backup.*` instruments. | +| `cluster_name` | Name of the PostgreSQL cluster the recording belongs to | All `klio.server.wal.*` instruments (counters, gauges, and the WAL duration histograms), `klio.client.wal.*`, the `klio.server.backup.*` PostgreSQL backup gauges (`backups`, `latest_backup_*`, `oldest_backup_*`) and the `klio.server.backup.relay` / `klio.server.backup.maintenance` counters. | +| `outcome` | `success`, `failure` | `klio.plugin.backup.runs`, `klio.server.backup.relay`, `klio.server.backup.maintenance`, `klio.server.backup.verifications`, and all WAL duration histograms (`klio.server.wal.*_duration`, `klio.client.wal.block_duration`). | +| `failure_category` | `repository_error`, `source_error`, `verification`, `timeout`, `canceled`, `unknown` | `klio.plugin.backup.runs` failure data points only. | +| `path` | `put` (WAL ingest), `get` (WAL serve) | `klio.server.wal.block_duration`, `klio.client.wal.block_duration`. | +| `stage` | put: `wrap`, `write`, `flush`, `send` (client); get: `read`, `unwrap`, `send` | `klio.server.wal.block_duration`, `klio.client.wal.block_duration`. | +| `snapshot_source` | Kopia source descriptor (`userName@hostName:path`) | All `klio.server.backup.*` base snapshot gauges (`snapshots`, `latest_snapshot_*`, `oldest_snapshot_timestamp`). | +| `stream` | JetStream stream name (`klio-wal-stream`, `klio-backup-stream`, `klio-latest-uploaded-wal-per-cluster-stream`) | `klio.server.queue.messages`, `klio.server.queue.bytes`. | + +### Backup lifecycle metrics (plugin sidecar) + +These metrics are emitted by the plugin sidecar and track backup +operations on each PostgreSQL instance: + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.plugin.backup.in_progress` | UpDownCounter | `{backups}` | Number of backups currently in progress | +| `klio.plugin.backup.latest_start_time` | Gauge | s | Unix epoch timestamp when the most recent backup started | +| `klio.plugin.backup.latest_completion_time` | Gauge | s | Unix epoch timestamp when the most recent backup completed successfully | +| `klio.plugin.backup.latest_failure_time` | Gauge | s | Unix epoch timestamp when the most recent backup failed | +| `klio.plugin.backup.latest_duration` | Gauge | s | Duration of the most recent backup | +| `klio.plugin.backup.duration` | Histogram | s | Distribution of backup durations, split by the `outcome` attribute (`success` / `failure`) | +| `klio.plugin.backup.runs` | Counter | `{backups}` | Total number of backup runs, split by the `outcome` attribute (`success` / `failure`). Failure data points additionally carry a `failure_category` attribute classifying the failure. Backup verification is part of a run: a verification failure is recorded here with `failure_category="verification"`, and a clean verification is included in the `outcome="success"` count | + +The `failure_category` attribute on `klio.plugin.backup.runs` failure +data points takes one of the following values: + +- `repository_error` — the backup failed while interacting with the + Klio server or the Kopia repository. +- `source_error` — the backup failed while connecting to or interacting + with the source PostgreSQL instance. +- `verification` — tier-1 verification detected corruption in the + freshly taken backup. +- `timeout` — the backup exceeded its deadline. +- `canceled` — the backup's context was canceled before a more specific + category could be determined. This covers cluster restart, + hibernation, pod eviction, and client disconnect; the metric does not + distinguish between them. +- `unknown` — the failure did not match any of the categories above. + +:::note +These metrics are tied to the plugin sidecar lifecycle: when the +sidecar restarts (for example, after a pod reschedule or PostgreSQL +instance failover) the counters reset to zero and the gauges are +re-initialized on the next backup. As a result, +`klio.plugin.backup.runs` reports totals since the last sidecar start +rather than over the life of the cluster, and may diverge from the +count of `Backup` resources. +::: + +### WAL ingest metrics (server) + +The WAL ingest series is unified across tiers: WAL bytes and files +written to local disk by the WAL gRPC server (tier 1) and uploaded +to remote storage by the consumer (tier 2) share a single instrument +family and are distinguished by the `tier` attribute (`"tier1"` or +`"tier2"`). + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.server.wal.written_size` | Counter | By | Number of bytes written for WAL files (per tier) | +| `klio.server.wal.written` | Counter | - | Number of WAL files written (per tier) | +| `klio.server.wal.latest_written_time` | Gauge | s | Unix epoch timestamp of the most recently written WAL file (per tier) | +| `klio.server.wal.latest_written_lsn` | Gauge | By | LSN of the most recently written WAL byte. On tier 1 this is the flush pointer (matches `pg_current_wal_flush_lsn()` semantics); on tier 2 this is the last byte of the most recently archived WAL segment | +| `klio.server.wal.latest_written_timeline` | Gauge | - | Timeline ID of the most recently completed WAL file (per tier) | + +Every recording carries a `cluster_name` attribute identifying the +PostgreSQL cluster, alongside the `tier` discriminator. + +### Post-backup processing metrics (server) + +The tier-1 backup itself is taken and counted client-side +(`klio.plugin.backup.runs`). Afterwards the server does two kinds of work for +the completed backup: optionally **relays** it to tier-2 (migration + +verification), and runs **maintenance** (base-snapshot retention + WAL +cleanup) on each tier. The relay is counted by `klio.server.backup.relay`; +maintenance is counted by `klio.server.backup.maintenance`, discriminated by a +`tier` attribute (`tier1` / `tier2`). + +Both carry `cluster_name` and `outcome` and are recorded once per attempt, so +a backup whose relay or maintenance is retried produces multiple data points +before it succeeds or is dead-lettered. + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.server.backup.relay` | Counter | `{relays}` | Number of tier-2 relay attempts after a backup (migration to tier-2 and verification), split by `cluster_name` and `outcome` (`success` / `failure`). | +| `klio.server.backup.maintenance` | Counter | `{runs}` | Number of maintenance runs after a backup (base-snapshot retention and WAL cleanup), split by `cluster_name`, `tier` (`tier1` / `tier2`) and `outcome` (`success` / `failure`) | + +### WAL duration histograms (server) + +The server records WAL processing latencies as OpenTelemetry +histograms. They replace the per-block spans Klio previously emitted +for each WAL stage, which were impractical for distributions: the +histograms can be aggregated across clusters and rendered as +percentile dashboards (`p50`, `p95`, `p99`) over time. Per-block +stages are recorded once per WAL block; the per-file instruments are +recorded once per WAL file. + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.server.wal.block_duration` | Histogram | ns | Per-block processing duration, split by the `path` (`put` ingest / `get` serve), `stage`, and `outcome` attributes. Put stages: `wrap`, `write`, `flush`; get stages: `read`, `unwrap`, `send`. Per-block send latency lives on the `send` stage (client `block_duration` for ingest, server `path="get"` for serve). Carries `tier` (`tier1` for put; `tier1` or `tier2` for get, depending on which WAL server handled it) and `cluster_name` | +| `klio.server.wal.get_duration` | Histogram | ns | Per-file duration of the gRPC get of a complete WAL file, split by `outcome`. Carries `tier` (`tier1` or `tier2`, depending on which WAL server served it) and `cluster_name` | +| `klio.server.wal.upload_duration` | Histogram | ns | Per-file duration of the tier-2 archival upload to remote storage, split by `outcome`. Carries `tier="tier2"` and `cluster_name` | + +The bucket boundaries are explicit (rather than an exponential +aggregation) so they survive export through the Prometheus bridge, and +are an initial set expected to be refined against real distributions. + +### WAL duration histograms (client) + +The WAL streaming client records the latency of shipping each WAL +block to the Klio server. + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.client.wal.block_duration` | Histogram | ns | Per-block duration of the gRPC send of a WAL block to the server. Carries `path="put"`, `stage="send"`, `cluster_name`, split by `outcome` | + +### WAL streaming state (client) + +The WAL streaming client also exposes the timeline it is currently +streaming. It is set when replication starts and updated on each +timeline switch (failover), giving a client-side, lag-free counterpart +to the server-side `klio.server.wal.latest_written_timeline` gauge +(which only advances once new-timeline WAL is written). + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.client.wal.timeline` | Gauge | - | Timeline ID the WAL streaming client is currently streaming. Carries `cluster_name` | + +### Backup verification metrics (server) + +As part of processing each backup the server verifies it. Verification +happens at two points: once against the tier-1 local copy, and again +against the tier-2 remote copy after migration. The `tier` attribute +(`"tier1"` or `"tier2"`) identifies which check the recording refers to. + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.server.backup.verifications` | Counter | `{verifications}` | Number of backup verifications, split by the `outcome` attribute (`success` / `failure`; `failure` indicates corruption detected) and the `tier` attribute | + +### Alerting on stalled WAL processing + +The same `klio.server.wal.latest_written_time` instrument is emitted +from two stages of the WAL pipeline, distinguished by the `tier` +attribute. A stale value signals a different failure depending on +the tier: + +- **`tier="tier1"`** reflects when the Klio server last received a WAL + file from PostgreSQL streaming replication and persisted it to + local disk. A stale value means PostgreSQL is no longer shipping + WALs to Klio, which may indicate a replication problem, or that + writing on disk is failing. + +- **`tier="tier2"`** reflects when the consumer last uploaded a WAL file + to tier-2 object storage. A stale value means the remote backend + is no longer receiving WALs, even though PostgreSQL replication + may still be working, or that uploading the WAL to the object store + is failing. + +### Server metrics + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.server.uptime` | Gauge | s | Klio server uptime in seconds | + +The `klio.server.wal.latest_written_lsn` instrument provides a +complementary view of the same two pipeline stages, expressed as a +byte offset rather than a wall-clock timestamp: + +- **`tier="tier1"`** is updated on every flushed WAL block received by + the WAL server (tracks `pg_current_wal_flush_lsn()` semantics). + +- **`tier="tier2"`** is updated once per completed WAL file by the + consumer. Its value is the LSN of the last byte of the WAL + segment just archived. + +The companion `klio.server.wal.latest_written_timeline` gauge +exposes the timeline ID of the WAL file each tier is currently +handling. + +:::warning +While usually increasing, the LSN gauge may decrease after the +promotion of a lagging standby. +::: + +Use these gauges alongside the timestamp gauges to distinguish a +slow pipeline (timestamps advancing, LSN gap growing) from a +stalled one (timestamps and LSN both frozen). + +### Base backup metrics (server) + +These metrics are emitted by the Klio server base backup component +and track Kopia snapshot statistics: + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.server.backup.snapshots` | Gauge | - | Total number of base snapshots | +| `klio.server.backup.latest_snapshot_size` | Gauge | By | Size of latest base snapshot in bytes (ignoring compression and deduplication) | +| `klio.server.backup.latest_snapshot_files` | Gauge | - | Number of files in latest base snapshot | +| `klio.server.backup.latest_snapshot_dirs` | Gauge | - | Number of directories in latest base snapshot | +| `klio.server.backup.latest_snapshot_timestamp` | Gauge | s | Unix epoch timestamp of the latest base snapshot | +| `klio.server.backup.oldest_snapshot_timestamp` | Gauge | s | Unix epoch timestamp of the oldest base snapshot | + +Every recording carries a `tier` attribute (`tier1` for the local +disk repository, `tier2` for the remote object store) and a +`snapshot_source` attribute identifying the source descriptor +(`userName@hostName:path`) the snapshot belongs to. + +The following metrics describe the retention window of physical +PostgreSQL backups, derived from the snapshotted backup metadata. +Each recording carries a `tier` attribute and a `cluster_name` +attribute identifying the PostgreSQL cluster the backup belongs to. +The `latest_backup_*` and `oldest_backup_*` gauges describe the most +recent and oldest backup retained on that tier (a base backup cannot +span a timeline switch, so its start and end share one timeline): + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.server.backup.backups` | Gauge | - | Number of PostgreSQL backups retained | +| `klio.server.backup.latest_backup_start_time` | Gauge | s | Unix epoch timestamp when the latest retained backup started | +| `klio.server.backup.latest_backup_completion_time` | Gauge | s | Unix epoch timestamp when the latest retained backup completed | +| `klio.server.backup.latest_backup_start_lsn` | Gauge | By | Start LSN of the latest retained backup (base 10) | +| `klio.server.backup.latest_backup_end_lsn` | Gauge | By | End LSN of the latest retained backup (base 10) | +| `klio.server.backup.latest_backup_timeline` | Gauge | - | Timeline of the latest retained backup | +| `klio.server.backup.oldest_backup_start_time` | Gauge | s | Unix epoch timestamp when the oldest retained backup started | +| `klio.server.backup.oldest_backup_completion_time` | Gauge | s | Unix epoch timestamp when the oldest retained backup completed | +| `klio.server.backup.oldest_backup_start_lsn` | Gauge | By | Start LSN of the oldest retained backup (base 10) | +| `klio.server.backup.oldest_backup_end_lsn` | Gauge | By | End LSN of the oldest retained backup (base 10) | +| `klio.server.backup.oldest_backup_timeline` | Gauge | - | Timeline of the oldest retained backup | + +### Queue metrics (server) + +These metrics are emitted by the Klio server and track the state of +the embedded NATS JetStream streams used for asynchronous Tier 2 +offloading of WAL files and backups. Each sample carries a `stream` +attribute identifying the source stream — typically +`klio-wal-stream` (WAL work queue), `klio-backup-stream` (backup work +queue), and `klio-latest-uploaded-wal-per-cluster-stream` (retention +safeguard, capped to one message per cluster): + +| Metric Name | Type | Unit | Description | +|---|---|---|---| +| `klio.server.queue.messages` | Gauge | - | Number of messages currently stored in the JetStream stream identified by `stream` | +| `klio.server.queue.bytes` | Gauge | By | Number of bytes currently stored in the JetStream stream identified by `stream` | + +### Migration from the previous metric names + +Klio is in alpha and previously emitted metrics under a flat +namespace. The component-based taxonomy above replaces those names in +a single hard rename — there is no dual emission. Update dashboards +and alerts according to the following table: + +| Previous name | New name | Notes | +|---|---|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `klio.backup.*` | `klio.plugin.backup.*` | Plugin sidecar metrics. | +| `klio.backup.running` | `klio.plugin.backup.in_progress` | Renamed and switched from a 0/1 gauge to an UpDownCounter; reports the number of concurrent backups in progress. | +| `klio.backup.latest_duration_seconds` | `klio.plugin.backup.latest_duration` | The `_seconds` suffix was dropped — the unit (`s`) is conveyed via the OpenTelemetry metric metadata, per [semantic conventions guidelines](https://opentelemetry.io/docs/specs/semconv/general/metrics/#units). The Prometheus export name is unchanged (`klio_plugin_backup_latest_duration_seconds`) because the Prometheus exporter appends the unit suffix when the OpenTelemetry name lacks it. | +| `klio.backup.successes`, `klio.backup.failures` | `klio.plugin.backup.runs` | Collapsed into a single counter with an `outcome` attribute (`success` / `failure`). | +| `klio.backup.verifications` | `klio.plugin.backup.runs` | Verification is part of a backup run; a clean verification is part of the `outcome="success"` count. | +| `klio.backup.verification_failures` | `klio.plugin.backup.runs{outcome="failure",failure_category="verification"}` | Verification corruption is recorded as a run failure with `failure_category="verification"`. | +| `klio.wal.written_size` | `klio.server.wal.written_size` | Carries `tier="tier1"`. | +| `klio.wal.written` | `klio.server.wal.written` | Carries `tier="tier1"`. | +| `klio.wal.latest_written_time` | `klio.server.wal.latest_written_time` | Carries `tier="tier1"`. | +| `klio.wal.latest_written_lsn` | `klio.server.wal.latest_written_lsn` | Carries `tier="tier1"`. | +| `klio.wal.latest_written_timeline` | `klio.server.wal.latest_written_timeline` | Carries `tier="tier1"`. | +| `klio.consumer.written_size` | `klio.server.wal.written_size` | Folded into the unified WAL series with `tier="tier2"`. | +| `klio.consumer.written` | `klio.server.wal.written` | Folded into the unified WAL series with `tier="tier2"`. | +| `klio.consumer.latest_written_time` | `klio.server.wal.latest_written_time` | Folded into the unified WAL series with `tier="tier2"`. | +| `klio.consumer.latest_written_lsn` | `klio.server.wal.latest_written_lsn` | Folded into the unified WAL series with `tier="tier2"`. | +| `klio.consumer.latest_written_timeline` | `klio.server.wal.latest_written_timeline` | Folded into the unified WAL series with `tier="tier2"`. | +| `klio.consumer.backup_verification_success` | `klio.server.backup.verifications{outcome="success"}` | Moved under `server.backup` to pair with `plugin.backup`; collapsed into a single counter with an `outcome` attribute. | +| `klio.consumer.backup_verification_failure` | `klio.server.backup.verifications{outcome="failure"}` | Same as above. | +| `klio.base.uptime` | `klio.server.uptime` | Server-level metric, not tied to Kopia. | +| `klio.base.*` | `klio.server.backup.*` | Snapshot metrics, folded under `server.backup` alongside the verification counters. | +| `klio.queue.*` | `klio.server.queue.*` | NATS JetStream metrics. Values are now reported per stream via a `stream` attribute instead of a single global aggregate. | + +## Configuration + +Klio automatically detects OpenTelemetry configuration through standard +environment variables. If no OpenTelemetry environment variables are present, +Klio will use no-op providers that don't collect any telemetry data. + +Traces and metrics exporters can be configured independently through the +[`autoexport`](https://go.opentelemetry.io/contrib/exporters/autoexport) package. + +### General Settings + +The following environment variables are used to configure OpenTelemetry: + +- `OTEL_SERVICE_NAME`: (required) Name of the service, e.g., `klio-server` +- `OTEL_RESOURCE_ATTRIBUTES`: Comma-separated list of resource attributes + (e.g., `deployment.environment=production,service.namespace=klio-system`) +- `OTEL_RESOURCE_DETECTORS`: Comma-separated list of resource detectors + from the [`autodetect`](https://pkg.go.dev/go.opentelemetry.io/contrib/detectors/autodetect) + package, used to automatically populate resource attributes + +### Traces exporter + +To enable the traces exporter, set the `OTEL_TRACES_EXPORTER` environment +variable to one of the supported exporters: + +- `otlp`: OpenTelemetry Protocol (OTLP) exporter +- `console`: Console exporter (useful for debugging) +- `none`: No-op exporter (disables tracing) + +You can define the OTLP protocol using the `OTEL_EXPORTER_OTLP_TRACES_PROTOCOL` +variable, or the general `OTEL_EXPORTER_OTLP_PROTOCOL`. Supported protocols +include: + +- `http/protobuf` (default) +- `grpc` + +Additional configuration options for trace exporters can be found in the +documentation of the respective exporters: + +- [OTLP Trace gRPC Exporter](https://pkg.go.dev/go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc) +- [OTLP Trace HTTP Exporter](https://pkg.go.dev/go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp) + +### Metrics Exporter + +To enable the metrics exporter, set the `OTEL_METRICS_EXPORTER` environment +variable to one of the supported exporters: + +- `otlp`: OpenTelemetry Protocol (OTLP) exporter +- `prometheus`: Prometheus exporter + HTTP server +- `console`: Console exporter (useful for debugging) +- `none`: No-op exporter (disables metrics) + +You can define the OTLP protocol using the `OTEL_EXPORTER_OTLP_METRICS_PROTOCOL` +variable, or the general `OTEL_EXPORTER_OTLP_PROTOCOL`. Supported protocols +include: + +- `http/protobuf` (default) +- `grpc` + +Additional configuration options for metrics exporters can be found in the +documentation of the respective exporters: + +- [OTLP Metric gRPC Exporter](https://pkg.go.dev/go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc) +- [OTLP Metric HTTP Exporter](https://pkg.go.dev/go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp) + +For the Prometheus exporter, you can configure the host and port of the HTTP +server using the following environment variables: + +- `OTEL_EXPORTER_PROMETHEUS_HOST` (default: `localhost`) +- `OTEL_EXPORTER_PROMETHEUS_PORT` (default: `9464`) + +### Exporters and receivers + +The OTLP exporter pushes telemetry to any OTLP-compatible receiver. Common +options include: + +- An [OpenTelemetry Collector](https://opentelemetry.io/docs/collector/), + which can receive OTLP data and fan it out to multiple backends + (Prometheus, Jaeger, Grafana, etc.). In Kubernetes, the + [OpenTelemetry Operator](https://opentelemetry.io/docs/platforms/kubernetes/operator/) + manages collectors via the `OpenTelemetryCollector` CRD and can expose + a stable in-cluster OTLP endpoint for Klio to target. +- Any backend with native OTLP support. + +The Prometheus exporter starts a local HTTP server that Prometheus scrapes +directly, with no intermediate collector required. + +## Configuring Klio with OpenTelemetry in Kubernetes + +When running in a Kubernetes environment, Klio will automatically define +`CONTAINER_NAME`, `POD_NAME` and `NAMESPACE_NAME` environment variables. +When any of these environment variables are set, Klio will automatically add +the corresponding resource attributes (`k8s.container.name`, `k8s.pod.name`, +`k8s.namespace.name`) to all of Klio's own telemetry (the server, the plugin +sidecars, the operator, and the WAL streaming client). Each attribute is added +independently - you don't need all three environment variables to be present. + +This applies to telemetry Klio emits itself. The Kopia subprocess exports its +own traces through a separate exporter that ignores these variables; see +[Kopia repository spans](#kopia-repository-spans) for how resource attributes +are populated on those spans. + +:::info +If you have already defined any of these attributes in +`OTEL_RESOURCE_ATTRIBUTES`, Klio will **not override** them. Only missing +attributes will be added from the environment variables. This allows you to +customize the values while still benefiting from automatic defaults for any +attributes you don't explicitly set. +::: + +### Klio server with OpenTelemetry + +When deploying a Klio `Server`, you can configure OpenTelemetry by specifying +the necessary settings in the `template` section of the `Server` spec: + +1. Set the required environment variables for OpenTelemetry configuration in + the `server` container. +1. Mount any necessary TLS certificates for secure communication with the + OpenTelemetry Collector. + +For simpler management, use a `ConfigMap` to store the OpenTelemetry configuration: + +```yaml +apiVersion: v1 +kind: ConfigMap +metadata: + name: klio-otel-config +data: + OTEL_SERVICE_NAME: "klio-server" + OTEL_RESOURCE_DETECTORS: "telemetry.sdk,host,os.type,process.executable.name" + OTEL_TRACES_EXPORTER: "otlp" + OTEL_EXPORTER_OTLP_TRACES_PROTOCOL: "grpc" + OTEL_EXPORTER_OTLP_TRACES_ENDPOINT: "https://otel-collector:4317" + OTEL_EXPORTER_OTLP_TRACES_COMPRESSION: "gzip" + OTEL_EXPORTER_OTLP_TRACES_TIMEOUT: "10000" + OTEL_EXPORTER_OTLP_TRACES_INSECURE: "false" + OTEL_EXPORTER_OTLP_TRACES_CERTIFICATE: "/otel/ca.crt" + OTEL_EXPORTER_OTLP_TRACES_CLIENT_CERTIFICATE: "/otel/tls.crt" + OTEL_EXPORTER_OTLP_TRACES_CLIENT_KEY: "/otel/tls.key" + OTEL_METRICS_EXPORTER: "otlp" + OTEL_METRIC_EXPORT_INTERVAL: "60000" + OTEL_EXPORTER_OTLP_METRICS_PROTOCOL: "grpc" + OTEL_EXPORTER_OTLP_METRICS_ENDPOINT: "https://otel-collector:4317" + OTEL_EXPORTER_OTLP_METRICS_TIMEOUT: "60000" + OTEL_EXPORTER_OTLP_METRICS_INSECURE: "false" + OTEL_EXPORTER_OTLP_METRICS_CERTIFICATE: "/otel/ca.crt" + OTEL_EXPORTER_OTLP_METRICS_CLIENT_CERTIFICATE: "/otel/tls.crt" + OTEL_EXPORTER_OTLP_METRICS_CLIENT_KEY: "/otel/tls.key" +--- +apiVersion: klio.cnpg.io/v1alpha1 +kind: Server +metadata: + name: my-klio-server +spec: + # ... other configuration ... + template: + spec: + containers: + - name: server + envFrom: + - configMapRef: + name: klio-otel-config + volumeMounts: + - mountPath: /otel + name: otel + volumes: + - name: otel + projected: + sources: + - secret: + name: otel-collector-tls + items: + - key: ca.crt + path: ca.crt + - secret: + name: otel-client-cert + items: + - key: tls.crt + path: tls.crt + - key: tls.key + path: tls.key +``` + +### Klio plugins with OpenTelemetry + +When deploying Klio as a CNPG Cluster plugin, configure OpenTelemetry by +specifying the necessary environment variables in the `containers` section of +the `PluginConfiguration` spec. The available container names are: + +- `klio-plugin`: Main plugin sidecar for backup management +- `klio-restore`: Restore operations sidecar + +Create a `ConfigMap` for the shared OpenTelemetry configuration: + +```yaml +apiVersion: v1 +kind: ConfigMap +metadata: + name: cluster-klio-otel-config +data: + OTEL_RESOURCE_DETECTORS: "telemetry.sdk,host,os.type,process.executable.name" + OTEL_TRACES_EXPORTER: "otlp" + OTEL_METRICS_EXPORTER: "otlp" + OTEL_EXPORTER_OTLP_PROTOCOL: "grpc" + OTEL_EXPORTER_OTLP_ENDPOINT: "https://otel-collector:4317" + OTEL_EXPORTER_OTLP_COMPRESSION: "gzip" + OTEL_EXPORTER_OTLP_TIMEOUT: "10000" + OTEL_EXPORTER_OTLP_INSECURE: "false" + OTEL_EXPORTER_OTLP_CERTIFICATE: "/projected/ca.crt" + OTEL_EXPORTER_OTLP_CLIENT_CERTIFICATE: "/projected/tls.crt" + OTEL_EXPORTER_OTLP_CLIENT_KEY: "/projected/tls.key" +``` + +Configure the `PluginConfiguration` to inject the environment variables into +each sidecar container: + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: PluginConfiguration +metadata: + name: client-config-cluster-example +spec: + serverAddress: klio.default + clientSecretName: cluster-example-klio-user + serverSecretName: klio-server-tls + clusterName: cluster-example + containers: + - name: klio-plugin + env: + - name: OTEL_SERVICE_NAME + value: "klio-plugin" + envFrom: + - configMapRef: + name: cluster-klio-otel-config + - name: klio-restore + env: + - name: OTEL_SERVICE_NAME + value: "klio-restore" + envFrom: + - configMapRef: + name: cluster-klio-otel-config +``` + +Mount the OpenTelemetry certificates using the Cluster's `projectedVolumeTemplate`. +The projected volume is mounted at `/projected/` and is accessible to all +sidecar containers: + +```yaml +apiVersion: postgresql.cnpg.io/v1 +kind: Cluster +metadata: + name: cluster-example +spec: + instances: 3 + + projectedVolumeTemplate: + sources: + - secret: + name: otel-collector-tls + items: + - key: ca.crt + path: ca.crt + - secret: + name: otel-client-cert + items: + - key: tls.crt + path: tls.crt + - key: tls.key + path: tls.key + + plugins: + - name: klio.cnpg.io + enabled: true + parameters: + pluginConfigurationRef: client-config-cluster-example + + storage: + size: 10Gi +``` + +### Klio operator with OpenTelemetry + +The operator bridges the controller-runtime Prometheus metrics +registry to OTLP and adds Go runtime and host instrumentation. +When no `OTEL_*` environment variables are present, a no-op +meter provider is installed and the operator runs without +telemetry overhead. + +The existing Prometheus `/metrics` endpoint remains available +for pull-based scraping regardless of whether OTLP export is +enabled. + +To enable OTLP export, set `OTEL_*` variables through the Helm +chart's `controllerManager.manager.env` value: + +```yaml +controllerManager: + manager: + env: + OTEL_SERVICE_NAME: "klio-operator" + OTEL_EXPORTER_OTLP_ENDPOINT: "http://otel-collector:4318" + OTEL_EXPORTER_OTLP_PROTOCOL: "http/protobuf" # or "grpc" +``` + +The Helm chart automatically injects `POD_NAME`, +`NAMESPACE_NAME`, and `CONTAINER_NAME` via the Kubernetes +downward API, so the corresponding `k8s.*` resource attributes +are populated without additional configuration. diff --git a/documentation/web/versioned_docs/version-0.0.18/user/plugin_configuration.md b/documentation/web/versioned_docs/version-0.0.18/user/plugin_configuration.md new file mode 100644 index 00000000..0ccde4ed --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/plugin_configuration.md @@ -0,0 +1,629 @@ +--- +sidebar_position: 6 +--- + +# The Klio Plugin + +The Klio plugin for CloudNativePG allows you to leverage the backup and WAL +streaming capabilities of Klio for your PostgreSQL clusters managed by +CloudNativePG. It adds a `klio-plugin` container to each PostgreSQL instance +pod, handling both backup creation/management and WAL streaming to the Klio +server in real-time. + +During recovery, Klio also injects a `klio-restore` container into the +CloudNativePG recovery Job to restore backups from the Klio server. See +[Available sidecar containers](#available-sidecar-containers) for details. + +## Configuration + +The Klio plugin integrates with CloudNativePG through the CNPG-I (CloudNativePG +Interface) specification, enabling Klio to manage backups and WAL streaming for +your PostgreSQL clusters. To use Klio with a CloudNativePG cluster, you need to: + +1. Create a `PluginConfiguration` resource that defines how to connect to the + Klio server +1. Reference the plugin in your `Cluster` resource specification + +## Prerequisites + +Before configuring a cluster to use the Klio plugin, ensure you have: + +- A running Klio `Server` resource deployed in your namespace +- A client TLS certificate stored in a Kubernetes Secret (see + [Klio server authentication](klio_server.md#authentication)) +- The server's TLS certificate available in a Secret + +## Creating a PluginConfiguration resource + +The `PluginConfiguration` custom resource defines how the Klio plugin connects +to and communicates with the Klio server. This resource contains connection +details, authentication credentials, and optional configuration for metrics, +profiling, and backup retention policies. + +### Basic example + +Here's a minimal `PluginConfiguration` example: + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: PluginConfiguration +metadata: + name: klio-plugin-config + namespace: default +spec: + serverAddress: klio-server.default + clientSecretName: client-sample-tls + serverSecretName: klio-server-tls + clusterName: my-cluster + # mode: standard # Optional: standard (default) or read-only +``` + +### Client credentials secret + +The client credentials must be stored in a Kubernetes Secret of type +`kubernetes.io/tls`, containing a secret to be presented to the Klio server. + +This secret can be generated with cert-manager by following the [documentation +in the Klio server page](klio_server.md#creating-a-client-side-certificate), +or be provided by the user. + +### Server Address + +The `serverAddress` field specifies where the Klio server can be reached. This +can be: + +- A Kubernetes service name: `klio-server.default` (within the same namespace) +- A fully qualified domain name: `klio-server.default.svc.cluster.local` +- An external address: `klio.example.com` + +Connections are made to the Klio server's two default listening ports: 51515 +for base backups and 52000 for WAL streaming. + +### TLS configuration + +The `serverSecretName` field references a Secret containing the TLS certificate +used to secure communication with the Klio server. This is the same +certificate configured on the `Server` resource. + +## Configuring a Cluster to use the Klio plugin + +Once you have created a `PluginConfiguration`, reference it in your CloudNativePG +`Cluster` resource: + +```yaml +apiVersion: postgresql.cnpg.io/v1 +kind: Cluster +metadata: + name: my-cluster + namespace: default +spec: + instances: 3 + + postgresql: + pg_hba: + - local replication all peer map=local # Allow replication connections locally + + plugins: + - name: klio.cnpg.io + enabled: true # Activate the Klio plugin (default) + parameters: + pluginConfigurationRef: klio-plugin-config + + storage: + size: 10Gi +``` + +To be able to stream WAL files, ensure that your PostgreSQL configuration +allows local replication connections. You can do this by adding an entry to the +`pg_hba` section, as shown in the example above. + +### Plugin parameters + +The `plugins` section in the `Cluster` specification requires: + +- **name**: Must be set to `klio.cnpg.io` to identify the Klio plugin +- **enabled**: Set to `true` to activate the plugin. This is the default value. +- **parameters.pluginConfigurationRef**: The name of your `PluginConfiguration` resource + +:::note +Even though the Klio plugin is used to archive WAL files on the Klio server, +it does not use the `archiveCommand` parameter in the PostgreSQL configuration, +as the WAL are streamed directly to the Klio server. Thus, you must not set +`isWALArchiver: true` in the plugin configuration. +::: + +:::info +If you add the Klio plugin to an **existing** cluster, you must +restart the cluster to inject the sidecar containers. Use +[`kubectl cnpg restart`][cnpg-restart] or set the +`kubectl.kubernetes.io/restartedAt` annotation on the cluster. + +[cnpg-restart]: https://cloudnative-pg.io/docs/current/kubectl-plugin/#restart +::: + +:::warning +The `PluginConfiguration` resource referenced in the `Cluster` should exist +before creating or updating the cluster. If it doesn't exist, the Klio plugin +uses the PreReconcile hook to gate reconciliation, causing the cluster to wait +at the start of the reconciliation loop (before any object creation or status +changes) until the `PluginConfiguration` is created. This avoids unrecoverable +error states, but the cluster will not progress until the dependency is +satisfied. Ensure the `PluginConfiguration` is created with the correct name. +::: + +## Applying configuration changes + +Most changes to the `PluginConfiguration` resource are applied +automatically. When you update a `PluginConfiguration`, the Klio +operator detects the change and updates the corresponding +`klio-config` Secret. The Klio sidecar containers monitor their +configuration file for changes and **restart automatically** to apply +the new configuration. This restart is intentional and expected. + +**What to expect during configuration updates:** + +- The `klio-plugin` sidecar container will restart to pick up the new + configuration +- Container restart counts will increment - this is normal behavior and + indicates the configuration was successfully applied +- Restarts are graceful and brief (typically 5-10 seconds) +- No data loss occurs - PostgreSQL continues running and WAL streaming + resumes automatically after restart +- You can verify the configuration was applied by checking the + `ConfigurationApplied` condition in the PluginConfiguration status + +This automatic propagation applies to all configuration fields +stored in the config file, including: + +- Server address +- Tier 1 and Tier 2 settings (backup, recovery, retention) +- Operation mode (`standard` or `read-only`) +- Cluster name override +- WAL prefetch configuration + +:::note +Changes to container customizations (such as `image`, +`resources`, or `securityContext`) and the `pprof` setting are +not applied automatically. These fields affect the pod spec, +which is managed by CloudNativePG. To apply these changes, use +`kubectl cnpg restart` to roll the cluster pods. +::: + +## Disabling the plugin + +To disable the Klio plugin for a CloudNativePG Cluster, set `enabled: false` +in the plugin reference inside the `Cluster` resource: + +```yaml +spec: + plugins: + - name: klio.cnpg.io + enabled: false + parameters: + pluginConfigurationRef: klio-plugin-config +``` + +After applying this change, a Pods rollout will be triggered by +CloudNativePG to remove the Klio sidecar containers. + +:::warning Replication slot cleanup +When the Klio plugin is active, the `klio-plugin` sidecar creates +a physical replication slot named `klio` on the PostgreSQL +primary. Disabling the plugin does **not** automatically remove +this replication slot. An orphaned replication slot will cause +PostgreSQL to retain WAL files indefinitely, eventually leading +to disk exhaustion. + +After disabling the plugin, connect to the primary instance and +drop the replication slot manually: + +```sql +SELECT pg_drop_replication_slot('klio'); +``` + +You can verify the slot exists beforehand with: + +```sql +SELECT slot_name, active +FROM pg_replication_slots +WHERE slot_type = 'physical'; +``` + +::: + +## Advanced configuration options + +The `PluginConfiguration` resource supports several advanced options to +customize the plugin's behavior. + +### Retention policies + +Define how long backups should be retained by configuring retention policies +for Tier 1 and Tier 2 storage. Retention policies can be configured +independently for each tier: + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: PluginConfiguration +metadata: + name: klio-plugin-config +spec: + serverAddress: klio-server.default + clientSecretName: klio-client-credentials + serverSecretName: klio-server-tls + clusterName: my-cluster + tier1: + retention: + keepLatest: 5 + keepHourly: 12 + keepDaily: 7 + keepWeekly: 4 + keepMonthly: 6 + keepAnnual: 2 + tier2: + enableBackup: true + enableRecovery: true + retention: + keepLatest: 10 + keepDaily: 30 + keepMonthly: 12 + keepAnnual: 5 +``` + +Except for `keepLatest`, each option defines how many backups to retain +for the specified time period. For example, `keepDaily: 7` means that we should +retain at most one backup for each of the past 7 days. + +If multiple backups exist within the same time bucket, the most recent one is +kept, unless preserved by a different *keep* rule. Backups that are not +retained by any rule are deleted. Rule evaluation is done when a new backup is +taken. + +The Klio server will automatically delete WAL files that are no longer needed +for recovery by any retained backup. + +All retention settings are optional. For each unspecified retention level, +the default Kopia value is applied: + +```yaml +keepLatest: 10 +keepHourly: 48 +keepDaily: 7 +keepWeekly: 4 +keepMonthly: 24 +keepAnnual: 1 +``` + +Set a rule to `0` to disable that retention level. + +### Operation Mode + +The `mode` field controls whether the plugin can perform both backup and +restore operations (`standard`) or only restore operations (`read-only`). + +:::note +The `mode` field is immutable. Once a PluginConfiguration is created, +its mode cannot be changed. To operate in a different mode, you would +need another PluginConfiguration with a different mode. +::: + +```yaml +spec: + mode: standard # or read-only +``` + +#### Standard Mode (default) + +In standard mode, the plugin can: + +- Create backups to Tier 1 (and optionally Tier 2) +- Stream WAL files to the server +- Restore from both Tier 1 and Tier 2 + +This is the default mode and suitable for most deployments. + +#### Read-Only Mode + +Use read-only mode when connecting to a read-only Klio server for recovery +operations only. This is useful for disaster recovery scenarios where you want +to restore from a read-only server in a secondary region or datacenter. + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: PluginConfiguration +metadata: + name: dr-restore-config +spec: + serverAddress: dr-server.default + clientSecretName: dr-client-credentials + serverSecretName: dr-server-tls + mode: read-only + # Must match the name of the original cluster whose backups you are + # restoring from, not the name of the new cluster being created. + clusterName: my-cluster + + # Read-only mode requires: + # - tier2 configuration with enableRecovery: true + # - NO tier1 configuration + tier2: + enableRecovery: true + # enableBackup must be false or omitted +``` + +**Restrictions in read-only mode:** + +- Tier 2 must be configured with `enableRecovery: true` +- Tier 2 `enableBackup` must be `false` or omitted +- Tier 1 configuration is not allowed +- No backup or WAL streaming operations are performed + +See the [Read-Only Server Mode](klio_server.md#read-only-mode) documentation +for details on setting up a read-only Klio server. + +### Cluster name override + +The `clusterName` field is required and tells Klio which PostgreSQL cluster's +backups and WAL archive to use. In most cases, set it to the name of the +CloudNativePG `Cluster` resource itself: + +```yaml +spec: + clusterName: my-cluster +``` + +Override it to a different value when the plugin needs to reference a +*different* cluster's data, for example when restoring into a new cluster +from an existing cluster's backups, or when configuring replica clusters: + +```yaml +spec: + clusterName: my-custom-cluster-name +``` + +:::warning Uniqueness across namespaces +`clusterName` is the sole identity key for a cluster's WAL/backup stream +on the Klio server it is registered with. It must be unique among all +clusters backed by the same Klio server, regardless of the Kubernetes +namespace they live in. Two clusters in different namespaces that share +a `clusterName` on the same server are not a supported configuration: +the second cluster to connect will have its WAL streaming rejected by +the server with an error such as: + +``` +Error: during server-side replication point validation: rpc error: code = InvalidArgument desc = invalid system ID, expected "" +``` + +This is expected, safe behavior: the server detects the mismatched +system ID and refuses to mix data from the two clusters. If you hit +this error, check whether another cluster on the same server is +already using the same `clusterName`. + +Deleting a cluster and later reusing its `clusterName` on the same +server hits the same error, since the original cluster backups and +WALs will still exist on the Klio server. +::: + +Whichever value you use, it must match the host name in the Common Name of the +client certificate (`userName@hostName`). For tier 1 base backups, a mismatch +is rejected at connection time; for WAL streaming, a mismatch is not currently +detected and can lead to WALs being stored or retrieved under the wrong +cluster path. + +### Tier 2 configuration + +Tier 2 provides secondary storage (typically object storage like S3) for +long-term backup retention and disaster recovery. Configure Tier 2 using the +`tier2` section: + +```yaml +spec: + tier2: + enableBackup: true + enableRecovery: true + retention: + keepDaily: 30 + keepMonthly: 12 +``` + +#### Options + +- **`enableBackup`**: When set to `true`, backups and WAL files are + automatically synchronized to Tier 2 storage after being stored in Tier 1. + This ensures your backups are available in long-term storage. + +- **`enableRecovery`**: When set to `true`, Klio will look for backups and + WAL files in both Tier 1 and Tier 2 during restore operations. If a backup + is available in both tiers, Tier 1 takes precedence as restore from it will + be faster. + +- **`retention`**: Configure a separate retention policy for Tier 2. + Typically, you would configure longer retention periods for Tier 2 since + object storage is more cost-effective for long-term storage. + +See the [Architecture documentation](./architectures.md#tier-2-secondary-storage-object-storage) +for more details on Tier 2 storage. + +### WAL Prefetch Configuration + +During recovery operations, Klio can download upcoming WAL files in the +background, ahead of PostgreSQL's requests, to speed up recovery. Configure +this behavior using the `walPrefetch` section: + +```yaml +spec: + walPrefetch: + count: 8 + maxConcurrentDownloads: 16 +``` + +#### Options + +- **`count`**: How many WAL files ahead of the current one to prefetch in + the background. Set to `0` to disable prefetching. Default is `2`. + +- **`maxConcurrentDownloads`**: The maximum number of WAL downloads — + prefetch and on-demand combined — that can run concurrently. Higher values + can improve recovery speed on high-bandwidth connections but use more + resources. Default is `4`. + +### Observability + +See the [OpenTelemetry observability](./opentelemetry.md) section for more +details on how to monitor the Klio plugin using OpenTelemetry. + +### Performance profiling + +Enable the pprof HTTP endpoint for performance profiling and troubleshooting: + +```yaml +spec: + pprof: true +``` + +When enabled, the pprof endpoint is exposed and can be used with Go's profiling +tools to analyze CPU usage, memory allocation, goroutines, and other runtime +metrics. + +:::warning +Only enable pprof in development or testing environments, or when actively +troubleshooting performance issues. It should not be enabled in production +unless necessary. +::: + +## Container customization + +The `PluginConfiguration` resource allows you to customize the Klio sidecar +containers by providing base container specifications that are used as the +foundation for the sidecars. This feature enables you to add custom environment +variables, volume mounts, resource limits, and other container settings without +modifying the PostgreSQL container environment. + +### Basic example + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: PluginConfiguration +metadata: + name: klio-plugin-config +spec: + serverAddress: klio-server.default + clientSecretName: klio-client-credentials + serverSecretName: klio-server-tls + clusterName: my-cluster + containers: + - name: klio-plugin + env: + - name: CUSTOM_ENV_VAR + value: "my-value" + - name: DEBUG_LEVEL + value: "info" +``` + +### How container merging works + +The containers you define serve as the base for the Klio sidecars, with the +following merge behavior: + +1. **Your container is the base**: When you define a container + (e.g., `klio-plugin`), your specification serves as the starting point +1. **Klio enforces required values**: Klio sets its essential configuration: + - Container `name` (klio-plugin or klio-restore) + - Container `args` (the command arguments needed for operation) + - `CONTAINER_NAME` environment variable +1. **Your customizations are preserved**: All other fields you define remain + intact +1. **Template defaults fill gaps**: For fields you don't specify, Klio applies + sensible defaults (image, security context, standard volume mounts, etc.) + +:::info + +Klio's required values (name, args, `CONTAINER_NAME` env var) will +always override any conflicting values you set. All other customizations are +respected. +::: + +### Plugin configuration selection in replica clusters + +When a CloudNativePG `Cluster` has multiple `PluginConfiguration` +resources — one for archiving and one or more for external clusters — +Klio selects which configuration to apply to the sidecar containers +using the following logic: + +1. **Archive plugin takes precedence**: if the `Cluster` has a Klio + plugin declared in `.spec.plugins` (the archive plugin), its + `PluginConfiguration` is used for the sidecar containers on every + pod. +1. **Fallback to replica source**: if no archive plugin is configured + and the cluster is a replica (`.spec.replica` is set), the + designated primary pod uses the `PluginConfiguration` referenced + by the external cluster defined as the replica source + (`.spec.replica.source`). Non-primary pods do not receive a Klio + sidecar in this case. + +For example, consider a replica cluster that does not archive locally +but restores WALs from an external source managed by Klio: + +```yaml +apiVersion: postgresql.cnpg.io/v1 +kind: Cluster +metadata: + name: cluster-dc-b +spec: + instances: 3 + # No archive plugin in .spec.plugins + + replica: + self: cluster-dc-b + primary: cluster-dc-a + source: cluster-dc-a + + externalClusters: + - name: cluster-dc-a + plugin: + name: klio.cnpg.io + parameters: + pluginConfigurationRef: source-plugin-config +``` + +In this setup, only the designated primary of `cluster-dc-b` gets a +Klio sidecar, configured from the `source-plugin-config` +`PluginConfiguration`. The container customizations defined in that +`PluginConfiguration` (image, resources, environment variables, etc.) +are applied to the sidecar following the same merging rules described +above. + +### Available sidecar containers + +The following containers can be customized: + +- **`klio-plugin`**: Handles backup creation/management and WAL streaming to + the Klio server in PostgreSQL instance pods +- **`klio-restore`**: Restores backups during recovery jobs + +### Example: Resource limits and environment variables + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: PluginConfiguration +metadata: + name: klio-plugin-config +spec: + serverAddress: klio-server.default + clientSecretName: klio-client-credentials + serverSecretName: klio-server-tls + clusterName: my-cluster + containers: + - name: klio-plugin + env: + - name: LOG_LEVEL + value: "debug" + - name: OTEL_EXPORTER_OTLP_ENDPOINT + value: "http://otel-collector:4317" + resources: + limits: + memory: "512Mi" + cpu: "1" + requests: + memory: "256Mi" + cpu: "500m" +``` diff --git a/documentation/web/versioned_docs/version-0.0.18/user/quickstart.md b/documentation/web/versioned_docs/version-0.0.18/user/quickstart.md new file mode 100644 index 00000000..bf803518 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/quickstart.md @@ -0,0 +1,463 @@ +--- +sidebar_position: 1.5 +--- + +# Quick Start + +This guide walks you through deploying a Klio server, configuring a +CloudNativePG cluster to back up to it, taking a backup, and restoring +it — end to end, on a single Tier 1 (local storage) server with +self-signed certificates. It's meant to get you to a working setup +quickly for evaluation purposes. + +For production topology decisions, see [Architectures](architectures.md). +For every option skipped here — Helm chart configuration, Tier 2 object +storage, read-only disaster-recovery servers, retention tuning, encryption +key rotation, IAM-based S3 authentication, container customization, and +more — see [Klio Operator Helm Chart](helm_chart.mdx), +[The Klio Server](klio_server.md), [The Klio Plugin](plugin_configuration.md), +and [Backup and Restore](backup_and_restore.md). + +## Prerequisites + +- A Kubernetes cluster with [CloudNativePG](https://cloudnative-pg.io) + installed +- `kubectl` configured to access your cluster +- [Helm](https://helm.sh/docs/intro/install/) installed (to install the + Klio operator) +- [cert-manager](https://cert-manager.io/) installed for certificate + management +- [Age](https://github.com/FiloSottile/age) CLI installed locally (for + encrypting the backup encryption key) +- Enough storage resources for the Klio server's data, cache, and queue + PersistentVolumeClaims + +## Install the Klio Operator + +Deploy the Klio Operator using its Helm chart, into the same namespace as +CloudNativePG (`cnpg-system` in these examples). This is a cluster-wide +component, separate from the `default` namespace used for the `Server` +and `Cluster` resources created later in this guide. + +If you don't have the Prometheus Operator installed, disable the +corresponding chart dependency: + +```yaml +# Uncomment if the Prometheus Operator is not installed +# prometheus: +# enable: false +``` + + +```sh +helm install klio-operator \ + oci://ghcr.io/cloudnative-pg/klio-operator-chart \ + --version 0.0.17 \ + --namespace cnpg-system \ + -f values.yaml +``` + + +Verify the operator is running and the CRDs were created: + +```sh +kubectl get pods -n cnpg-system -l app.kubernetes.io/name=klio +kubectl get crds | grep klio.cnpg.io +``` + +For the full configuration reference, upgrade procedure, and uninstall +instructions, see [Klio Operator Helm Chart](helm_chart.mdx). + +## Step 1: Deploy a Klio Server + +### 1.1 Create the encryption key + +The encryption key protects backup data at rest. Klio uses +[Age](https://github.com/FiloSottile/age) encryption to protect the key +itself, so it can be rotated without touching the Kopia repository. + +Generate an Age key pair: + +```bash +age-keygen -o identity.txt +# Public key: age1ql3z7hjy54pw3hyww5ayyfg7zqgvc7w3j2elw8zmrj2kg5sfn9aqmcac8p +``` + +Generate a random encryption key and encrypt it with the public key: + +```bash +openssl rand -hex 32 | age \ + -r age1ql3z7hjy54pw3hyww5ayyfg7zqgvc7w3j2elw8zmrj2kg5sfn9aqmcac8p \ + -o encryption-key.age +``` + +Create Kubernetes Secrets for both files: + +```bash +kubectl create secret generic my-server-encryption-key \ + --from-file=encryption-key.age +kubectl create secret generic my-server-age-identity \ + --from-file=identity.txt +``` + +:::tip +Use a strong, randomly generated key. Store it securely — there is no key +recovery mechanism. +::: + +### 1.2 Create a CA certificate + +The CA is used to authenticate Klio clients (the plugin sidecars) via +mTLS. Using cert-manager, create a self-signed CA: + +```yaml +--- +apiVersion: cert-manager.io/v1 +kind: Issuer +metadata: + name: selfsigned-issuer + namespace: default +spec: + selfSigned: { } +--- +apiVersion: cert-manager.io/v1 +kind: Certificate +metadata: + name: server-sample-ca +spec: + commonName: server-sample-ca + secretName: server-sample-ca + + duration: 2160h # 90d + renewBefore: 360h # 15d + + isCA: true + usages: + - cert sign + + issuerRef: + name: selfsigned-issuer + kind: Issuer + group: cert-manager.io +``` + +```bash +kubectl apply -f ca-configuration.yaml +``` + +### 1.3 Create the server's TLS certificate + +```yaml +--- +apiVersion: cert-manager.io/v1 +kind: Issuer +metadata: + name: selfsigned-issuer + namespace: default +spec: + selfSigned: { } +--- +apiVersion: cert-manager.io/v1 +kind: Certificate +metadata: + name: my-server-cert + namespace: default +spec: + secretName: my-server-tls + commonName: my-server + dnsNames: + - my-server + - my-server.default + - my-server.default.svc + - my-server.default.svc.cluster.local + duration: 2160h # 90 days + renewBefore: 360h # 15 days + isCA: false + usages: + - server auth + issuerRef: + name: selfsigned-issuer + kind: Issuer + group: cert-manager.io +``` + +```bash +kubectl apply -f tls-certificate.yaml +``` + +:::info +For production environments, use certificates signed by your +organization's Certificate Authority (CA) or a trusted public CA instead +of self-signed certificates. +::: + +### 1.4 Create the Server resource + + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: Server +metadata: + name: my-server + namespace: default +spec: + # Container image for the Klio server + image: ghcr.io/cloudnative-pg/klio:v0.0.17 + imagePullPolicy: IfNotPresent + + # TLS configuration + tlsSecretName: my-server-tls + + # Client authentication configuration + caSecretName: server-sample-ca + + # tier 1 configuration + tier1: + # Cache storage configuration + cache: + pvcTemplate: + storageClassName: standard # Adjust to your storage class (use 'kubectl get storageclass' to see available options) + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 10Gi # Adjust based on your needs + # Data storage pvcTemplate (for backups and WAL) + data: + pvcTemplate: + storageClassName: standard # Adjust to your storage class (use 'kubectl get storageclass' to see available options) + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 100Gi # Adjust based on your backup needs + # Age-encrypted encryption key file + encryptionKeyFile: + fileReference: + volume: + secret: + secretName: my-server-encryption-key + path: encryption-key.age + # Age identity file for decryption + identityFile: + fileReference: + volume: + secret: + secretName: my-server-age-identity + path: identity.txt + + # Queue storage configuration (for NATS work queue) + # Required when tier1 is configured + queue: + pvcTemplate: + storageClassName: standard # Adjust to your storage class + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 50Mi +``` + + +```bash +kubectl apply -f klio-server.yaml +``` + +### 1.5 Verify the server is running + +```bash +# Check the Server resource status +kubectl get server my-server -n default + +# Check the StatefulSet +kubectl get statefulset my-server-klio -n default + +# Check the Pod +kubectl get pods -l klio.cnpg.io/klio-server=my-server -n default + +# View logs +kubectl logs -l klio.cnpg.io/klio-server=my-server -n default -f +``` + +The server should create a StatefulSet with a pod named `my-server-klio-0`. + +## Step 2: Configure a Cluster with the Klio Plugin + +### 2.1 Create a client certificate + +This certificate authenticates the PostgreSQL cluster's Klio sidecar to +the server. The `commonName` must follow the `@` +format, where `` matches the `clusterName` you'll set on the +`PluginConfiguration` below. + +```yaml +apiVersion: cert-manager.io/v1 +kind: Certificate +metadata: + name: client-sample-tls +spec: + secretName: client-sample-tls + commonName: klio@my-cluster + + duration: 2160h # 90d + renewBefore: 360h # 15d + + isCA: false + usages: + - client auth + + issuerRef: + name: server-sample-ca + kind: Issuer + group: cert-manager.io +``` + +```bash +kubectl apply -f client-certificate.yaml +``` + +### 2.2 Create a PluginConfiguration + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: PluginConfiguration +metadata: + name: klio-plugin-config + namespace: default +spec: + serverAddress: my-server.default + clientSecretName: client-sample-tls + serverSecretName: my-server-tls + clusterName: my-cluster +``` + +```bash +kubectl apply -f klio-plugin-config.yaml +``` + +### 2.3 Reference the plugin in your Cluster + +```yaml +apiVersion: postgresql.cnpg.io/v1 +kind: Cluster +metadata: + name: my-cluster + namespace: default +spec: + instances: 3 + + postgresql: + pg_hba: + - local replication all peer map=local # Allow replication connections locally + + plugins: + - name: klio.cnpg.io + enabled: true # Activate the Klio plugin (default) + parameters: + pluginConfigurationRef: klio-plugin-config + + storage: + size: 10Gi +``` + +```bash +kubectl apply -f my-cluster.yaml +``` + +The `pg_hba` entry above is required so PostgreSQL accepts the local +replication connection the Klio plugin uses to stream WAL files. + +## Step 3: Take a Backup + +```yaml +apiVersion: postgresql.cnpg.io/v1 +kind: Backup +metadata: + name: my-cluster-backup-20251027 + namespace: default +spec: + method: plugin + target: primary + cluster: + name: my-cluster + pluginConfiguration: + name: klio.cnpg.io +``` + +```bash +kubectl apply -f backup.yaml +``` + +See [Monitor Backup Progress](backup_and_restore.md#monitor-backup-progress) +to check on it. + +## Step 4: Restore the Backup + +Restoring bootstraps a **new** cluster from the backup. First, create a +`PluginConfiguration` pointing back at the original cluster's data: + +```yaml +apiVersion: klio.cnpg.io/v1alpha1 +kind: PluginConfiguration +metadata: + name: my-restore-config + namespace: default +spec: + serverAddress: my-server.default + clientSecretName: client-sample-tls + serverSecretName: my-server-tls + # Must match the name of the original cluster whose backups you are + # restoring from, not the name of the new cluster being created. + clusterName: my-cluster +``` + +```bash +kubectl apply -f restore-config.yaml +``` + +Then create the restored `Cluster`: + +```yaml +apiVersion: postgresql.cnpg.io/v1 +kind: Cluster +metadata: + name: my-restored-cluster + namespace: default +spec: + instances: 3 + + # Bootstrap from a Klio backup + bootstrap: + recovery: + source: source + # OPTIONAL: Specify the backup to restore from; Klio picks the + # latest one automatically if omitted + recoveryTarget: + backupID: my-cluster-backup-YYYYMMDDHHMMSS + + # Reference the Klio plugin configuration + externalClusters: + - name: source + plugin: + name: klio.cnpg.io + parameters: + pluginConfigurationRef: my-restore-config + + storage: + size: 10Gi +``` + +```bash +kubectl apply -f restored-cluster.yaml +``` + +The restored cluster operates independently and will **not** perform its +own backups unless you configure the Klio plugin for backup operations on +it too, as in Step 2. + +## Next Steps + +- [Architectures](architectures.md) — production topology options, + including Tier 2 object storage and read-only disaster-recovery servers +- [Managing Storage](managing_storage.md) — sizing and resizing PVCs +- [OpenTelemetry](opentelemetry.md) — monitoring and telemetry diff --git a/documentation/web/versioned_docs/version-0.0.18/user/upgrade_notes.md b/documentation/web/versioned_docs/version-0.0.18/user/upgrade_notes.md new file mode 100644 index 00000000..aaf16d01 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/upgrade_notes.md @@ -0,0 +1,9 @@ +--- +sidebar_position: 91 +--- + +# Upgrade Notes + +This page lists version-specific changes that may require +manual action when upgrading Klio. For the upgrade procedure, +see the [Helm chart page](helm_chart.mdx#upgrades). diff --git a/documentation/web/versioned_docs/version-0.0.18/user/wal_streaming.md b/documentation/web/versioned_docs/version-0.0.18/user/wal_streaming.md new file mode 100644 index 00000000..4734b361 --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/wal_streaming.md @@ -0,0 +1,120 @@ +--- +sidebar_position: 4 +--- + +# WAL Streaming + +A standout feature of Klio is its native, cloud-first implementation of WAL +streaming for PostgreSQL. This architecture enables: + +- Partial WAL segment streaming, ensuring real-time data transfer +- Built-in compression and encryption using user-provided keys +- Controlled replication slot advancement, protecting against WAL loss +- Optional synchronous replication, offering zero RPO when enabled + +## Architecture + +WAL streaming in Klio is built around two components: a client and a server. + +- The client, invoked using the `klio send-wal` command, typically runs + alongside PostgreSQL but does not have to. +- The server, started with the `klio server start` command, runs as a + dedicated process on the Klio server. + +In Kubernetes environments, as illustrated in the diagram below, Klio streams +WAL records directly from the PostgreSQL primary over a local Unix domain +socket. The WAL streamer runs as a lightweight sidecar container within the +same pod as the primary instance and is managed by the CNPG-I–compliant plugin. +It continuously pushes data to a remote Klio WAL server (Tier 1), which handles +partial WAL file synchronization and archives completed segments into the +central WAL archive for the PostgreSQL cluster. + +![WAL streaming architectural overview](images/wal-streaming.png) + +## Moving Beyond `archive_command` + +Klio replaces the traditional PostgreSQL `archive_command` method for WAL +handling in CloudNativePG clusters, providing improved reliability, efficiency, +security, and observability. + +PostgreSQL’s `archive_command` is a shell command executed when a WAL segment +is complete—either because the segment reached its size limit (typically 16MB) +or the `archive_timeout` elapsed (5 minutes by default in CloudNativePG). + +The streaming model provided by Klio offers several key advantages over this +approach: + +- **Near-zero RPO:** WAL changes are streamed incrementally in near real-time, + reducing the worst-case recovery point objective (RPO) from 5 minutes to + near-zero, or even zero in synchronous mode. + +- **Improved efficiency and scalability:** A single, continuously running WAL + streamer process replaces the need to spawn a new process for each WAL + segment, resulting in lower CPU and I/O usage and better scalability during + periods of high WAL volume. + +- **Enhanced security:** WAL data is encrypted end-to-end, both in transit and + at rest, providing protection not available with the traditional + `archive_command`. + +- **Comprehensive observability:** Native metrics and structured logging + provide full visibility into WAL streaming operations, simplifying + monitoring, anomaly detection, and troubleshooting compared to the opaque + nature of `archive_command`. + +## Monitoring Klio WAL Streamer in PostgreSQL + +The Klio WAL streamer is a PostgreSQL streaming replication client and, +as such, can be monitored using the standard `pg_stat_replication` +system view in the PostgreSQL catalog. + +The WAL streamer identifies itself with `application_name` set to `klio`. + +To verify whether any Klio WAL streamer is connected to an instance (in +Kubernetes deployments, this will always be the primary), run the following +query: + +```sql +SELECT * FROM pg_stat_replication WHERE application_name = 'klio'; +``` + +An example output might look like this: + +The following excerpt is an a example: + +```console +-[ RECORD 1 ]----+------------------------------ +pid | 1070 +usesysid | 10 +usename | postgres +application_name | klio +client_addr | +client_hostname | +client_port | -1 +backend_start | 2025-08-07 01:14:39.619662+00 +backend_xmin | +state | streaming +sent_lsn | 2/C765A000 +write_lsn | 2/C75FA000 +flush_lsn | 2/C741A000 +replay_lsn | 2/C741A000 +write_lag | 00:00:00.919907 +flush_lag | 00:00:00.923556 +replay_lag | 00:00:00.923556 +sync_priority | 0 +sync_state | async +reply_time | 2025-08-07 01:54:44.756306+00 +``` + +As you can see, Klio provides relevant feedback to PostgreSQL. Here is a brief +explanation of the key fields: + +- `state`: The replication connection status (`streaming` indicates active + streaming). +- `sent_lsn`, `write_lsn`, `flush_lsn`, `replay_lsn`: Positions in the WAL + indicating how far data has been sent, written, flushed, and replayed on the + Klio server (replayed and flushed are always identical). +- `write_lag`, `flush_lag`, `replay_lag`: Delays between WAL positions + indicating replication latency. +- `sync_state`: The synchronization state of this standby (e.g., `async`, + `sync`, `potential`, `quorum`). diff --git a/documentation/web/versioned_docs/version-0.0.18/user/walplayer.md b/documentation/web/versioned_docs/version-0.0.18/user/walplayer.md new file mode 100644 index 00000000..2e888e6e --- /dev/null +++ b/documentation/web/versioned_docs/version-0.0.18/user/walplayer.md @@ -0,0 +1,335 @@ +--- +title: WAL Player +sidebar_position: 80 +--- + +# WAL Player + +The WAL Player is a command-line tool designed to benchmark the performance of +your Klio servers by simulating PostgreSQL Write-Ahead Log (WAL) file streaming +workloads. It is an essential tool for ensuring your Klio servers can handle +your production workloads efficiently. Use it regularly to validate performance +and capacity planning decisions. + +## Overview + +WAL Player provides two main commands: + +- **`generate`** - Creates synthetic WAL files for testing +- **`play`** - Sends WAL files to a Klio server and measures performance + +This tool is essential for: + +- Performance testing and benchmarking Klio servers +- Validating server capacity under different workloads +- Measuring throughput and latency characteristics +- Load testing before production deployment + +## Prerequisites + +- Klio binary installed and accessible +- A running Klio server to test against +- Sufficient disk space for generating test WAL files + +## Commands + +### `klio wal-player generate` + +Generates synthetic WAL files for testing purposes. + +#### Usage + +```bash +klio wal-player generate [output-directory] [flags] +``` + +#### Parameters + +- `output-directory` - Directory where WAL files will be created (defaults to + current directory) + +#### Flags + +- `--wal-size` - Size of each WAL file in MB (default: 16) +- `--length` - Number of WAL files to generate (required) + +#### Examples + +```bash +# Generate 10 WAL files of 16MB each in the current directory +klio wal-player generate --length 10 + +# Generate 50 WAL files of 32MB each in a specific directory +klio wal-player generate /tmp/test-wals --wal-size 32 --length 50 +``` + +### `klio wal-player play` + +Sends WAL files to a Klio server and measures performance metrics. + +#### Usage + +```bash +klio wal-player play [directory] [flags] +``` + +#### Parameters + +- `directory` - Directory containing WAL files to send (required). This + directory should contain PostgreSQL WAL files in the standard format (e.g., + `000000010000000000000001`). It also supports files compressed with gzip, + provided they have the `.gz` extension. + +#### Flags + +- `--jobs, -j` - Number of parallel jobs for concurrent uploads (default: 1). + Can be used to simulate multiple Klio clients sending data simultaneously. +- `--block-size` - Block size in KB for streaming (default: 2048). This controls + how much data is sent in each request. + +#### Configuration + +The play command requires client configuration to connect to your Klio server. +This can be provided via: + +- Configuration file +- Environment variables +- Command-line flags + +Example configuration: + +```yaml +# klio-config.yaml +client: + cluster_name: walplayer + wal: + address: localhost:52000 + server_cert_path: "/path/to/server.crt" + client_cert_path: "/path/to/client/tls.crt" + client_key_path: "/path/to/client/tls.key" +``` + +#### Examples + +```bash +# Send WAL files using single connection +klio wal-player play ./test-wals + +# Send WAL files using 4 workers for parallel uploads +klio wal-player play ./test-wals --jobs 4 + +# Benchmark with different block sizes +klio wal-player play ./test-wals --jobs 2 --block-size 1024 +``` + +## Performance Metrics + +The `play` command outputs detailed performance metrics in JSON format for each +WAL file: + +```json +{ + "walFullPath": "/path/to/000000010000000000000001", + "startTime": "2025-01-15T10:30:00Z", + "endTime": "2025-01-15T10:30:02Z", + "elapsedTime": "7651680", + "error": "" +} +``` + +### Metrics Explained + +- **`walFullPath`** - Full path to the WAL file that was sent +- **`startTime`** - When the upload started +- **`endTime`** - When the upload completed +- **`elapsedTime`** - Total time taken for the upload in nanoseconds +- **`error`** - Error message if the upload failed (empty on success) + +## Benchmarking Example + +The following Kubernetes Job definition demonstrates how to use +the WAL Player to benchmark a Klio server. This example covers generating +WAL files and then playing them back to the server. + + +```yaml +--- +# PVC for storing generated WAL files +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: walplayer-data +spec: + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 2Gi # Enough space to hold the generated amount of WAL files +--- +# Client certificate for authenticating with the Klio server +apiVersion: cert-manager.io/v1 +kind: Certificate +metadata: + name: walplayer-client-cert +spec: + commonName: klio@walplayer + secretName: walplayer-client-cert + duration: 2160h # 90d + renewBefore: 360h # 15d + isCA: false + usages: + - client auth + issuerRef: + name: server-sample-ca + kind: Issuer + group: cert-manager.io +--- +# ConfigMap with klio client configuration +apiVersion: v1 +kind: ConfigMap +metadata: + name: walplayer-config +data: + # Address your klio server + klio-config.yaml: | + client: + cluster_name: walplayer + wal: + address: server-sample.default:52000 + server_cert_path: /certs/server/ca.crt + client_cert_path: /certs/client/tls.crt + client_key_path: /certs/client/tls.key +--- +# Job to generate and play WAL files +apiVersion: batch/v1 +kind: Job +metadata: + name: walplayer-benchmark +spec: + template: + metadata: + labels: + app: walplayer + spec: + restartPolicy: Never + initContainers: + # Generate synthetic WAL files + - name: generate-wals + image: ghcr.io/cloudnative-pg/klio:v0.0.18 + imagePullPolicy: Always + command: + - /usr/bin/klio + - wal-player + - generate + - /data + - --wal-size=16 + - --length=100 + volumeMounts: + - name: data + mountPath: /data + containers: + # Play WAL files to the Klio server + - name: play-wals + image: ghcr.io/cloudnative-pg/klio:v0.0.18 + imagePullPolicy: Always + command: + - /usr/bin/klio + - wal-player + - play + - /data + - --config=/config/klio-config.yaml + - --jobs=4 + - --block-size=2048 + volumeMounts: + - name: data + mountPath: /data + - name: config + mountPath: /config + readOnly: true + - name: server-cert + mountPath: /certs/server + readOnly: true + - name: client-cert + mountPath: /certs/client + readOnly: true + volumes: + - name: data + persistentVolumeClaim: + claimName: walplayer-data + - name: config + configMap: + name: walplayer-config + - name: server-cert + secret: + secretName: server-sample-tls + - name: client-cert + secret: + secretName: walplayer-client-cert +``` + + +### Customizing the Benchmark + +You can adjust the following parameters to simulate different workload scenarios: + +#### WAL Generation Parameters + +Modify the `generate-wals` init container to create different test workloads: + +- Many small files: + + ```yaml + - --wal-size=16 + - --length=1000 + ``` + +- Less large files: + + ```yaml + - --wal-size=256 + - --length=100 + ``` + +Match actual production for better results. + +#### WAL Playback Parameters + +Modify the `play-wals` container to test different upload patterns: + +- **Jobs** (`--jobs`): Number of parallel upload workers + - Start with `--jobs=1` to establish baseline performance + - Increase to `--jobs=2`, `--jobs=4`, `--jobs=8` to find optimal concurrency + - Performance typically plateaus at some point +- **Block Size** (`--block-size`): Size of each streaming chunk in KB + - Default is `--block-size=2048` + - Maximum is 8192 + +### Analyzing Results + +View the job logs to see the JSON performance metrics: + +```bash +kubectl logs job/walplayer-benchmark -c play-wals +``` + +You can analyze the output using `jq`: + +```bash +# Get all results +kubectl logs job/walplayer-benchmark -c play-wals > results.json + +# Calculate total successful uploads +jq -s '[.[] | select(.error == "")] | length' results.json + +# Calculate average upload time (in nanoseconds) +jq -s '[.[] | select(.error == "") | .elapsedTime | tonumber] | add / length' results.json + +# Find any failed uploads +jq -s '.[] | select(has("error") and .error != "")' results.json +``` + +## Performance Optimization Tips + +1. **Resource Monitoring**: Monitor CPU, memory, and disk I/O on the Klio server +1. **Network Bandwidth**: Ensure sufficient bandwidth between client and server +1. **Storage Performance**: Verify storage can handle the write throughput diff --git a/documentation/web/versioned_sidebars/version-0.0.18-sidebars.json b/documentation/web/versioned_sidebars/version-0.0.18-sidebars.json new file mode 100644 index 00000000..1ed7b9fe --- /dev/null +++ b/documentation/web/versioned_sidebars/version-0.0.18-sidebars.json @@ -0,0 +1,14 @@ +{ + "docs": [ + { + "type": "autogenerated", + "dirName": "user" + } + ], + "developers_docs": [ + { + "type": "autogenerated", + "dirName": "developer" + } + ] +} diff --git a/documentation/web/versions.json b/documentation/web/versions.json new file mode 100644 index 00000000..ac8d9edf --- /dev/null +++ b/documentation/web/versions.json @@ -0,0 +1,3 @@ +[ + "0.0.18" +] diff --git a/operator/dist/chart/Chart.yaml b/operator/dist/chart/Chart.yaml index 877e5ff1..68f08c18 100644 --- a/operator/dist/chart/Chart.yaml +++ b/operator/dist/chart/Chart.yaml @@ -10,8 +10,8 @@ keywords: - backup - CloudNativePG type: application -version: 0.0.17 -appVersion: 0.0.17 +version: 0.0.18 +appVersion: 0.0.18 home: https://github.com/cloudnative-pg/klio maintainers: - name: fcanovai diff --git a/operator/dist/chart/values.yaml b/operator/dist/chart/values.yaml index e61d9e8a..3f4ad388 100644 --- a/operator/dist/chart/values.yaml +++ b/operator/dist/chart/values.yaml @@ -24,7 +24,7 @@ controllerManager: # -- The image to use for the controller manager container. repository: ghcr.io/cloudnative-pg/klio-operator # -- The tag to use for the controller manager container image. - tag: v0.0.17 # x-release-please-version + tag: v0.0.18 # x-release-please-version # -- The controller manager container imagePullPolicy. pullPolicy: Always # -- The list of imagePullSecrets. @@ -33,7 +33,7 @@ controllerManager: # Set OTEL_* variables here to enable OpenTelemetry export. # See the OpenTelemetry Observability page in the documentation. env: { - SIDECAR_IMAGE: "ghcr.io/cloudnative-pg/klio:v0.0.17" # x-release-please-version + SIDECAR_IMAGE: "ghcr.io/cloudnative-pg/klio:v0.0.18" # x-release-please-version } # -- Liveness probe configuration. livenessProbe: diff --git a/operator/internal/cnpgi/metadata.go b/operator/internal/cnpgi/metadata.go index d870dfdf..8a0baea2 100644 --- a/operator/internal/cnpgi/metadata.go +++ b/operator/internal/cnpgi/metadata.go @@ -28,7 +28,7 @@ import ( // data is the metadata of this plugin. var data = identity.GetPluginMetadataResponse{ //nolint: gochecknoglobals Name: klioconfig.PluginName, - Version: "0.0.17", // x-release-please-version + Version: "0.0.18", // x-release-please-version DisplayName: "Klio", ProjectUrl: "", RepositoryUrl: "",