diff --git a/charts/paradedb/docs/runbooks/CNPGClusterLogicalReplicationDistance.md b/charts/paradedb/docs/runbooks/CNPGClusterLogicalReplicationDistance.md new file mode 100644 index 0000000000..df0e57c885 --- /dev/null +++ b/charts/paradedb/docs/runbooks/CNPGClusterLogicalReplicationDistance.md @@ -0,0 +1,34 @@ +# CNPGClusterLogicalReplicationDistance + +## Description + +These alerts report the subscriber's `received_lsn` minus `latest_end_lsn`. +Warning fires above 1 GiB for five minutes; critical fires above 4 GiB for two +minutes. Labels identify the pod, database, and subscription. + +## Impact + +A large WAL-position gap warrants investigation but does not measure unapplied +work or committed apply progress. Missing worker metrics cannot trigger these +alerts; check worker health separately. + +## Diagnosis + +Inspect the subscription on the subscriber database: + +```sql +SELECT subname, pid, received_lsn, latest_end_lsn, + last_msg_receipt_time, latest_end_time, + pg_wal_lsn_diff(received_lsn, latest_end_lsn) AS position_gap_bytes +FROM pg_stat_subscription; +``` + +Check receipt age and worker health, and use the +[logical replication error runbook](CNPGClusterLogicalReplicationErrors.md) to +inspect apply/sync errors and subscriber logs. + +## Mitigation + +Address any confirmed connection, worker, or apply errors found during diagnosis. +Do not skip transactions or resynchronize a subscription based on this position +gap alone. diff --git a/charts/paradedb/prometheus_rules/cluster-logical_replication_distance-critical.yaml b/charts/paradedb/prometheus_rules/cluster-logical_replication_distance-critical.yaml new file mode 100644 index 0000000000..ec5f3adf84 --- /dev/null +++ b/charts/paradedb/prometheus_rules/cluster-logical_replication_distance-critical.yaml @@ -0,0 +1,25 @@ +{{- $alert := "CNPGClusterLogicalReplicationDistanceCritical" -}} +{{- if not (has $alert .excludeRules) -}} +alert: {{ $alert }} +annotations: + summary: ParadeDB logical replication WAL-position distance is critical + description: |- + ParadeDB CNPG Cluster "{{ .namespace }}/{{ .cluster }}" reports a WAL-position difference of {{ .value }} GiB as the logical replication subscriber for subscription "{{ .labels.subname }}" in database "{{ .labels.datname }}". + + This does not measure unapplied data. Check replication errors, worker health, and subscriber logs. + runbook_url: https://github.com/paradedb/charts/blob/main/charts/paradedb/docs/runbooks/CNPGClusterLogicalReplicationDistance.md +expr: | + max by (namespace, job, pod, datname, subname) ( + cnpg_pg_stat_subscription_received_lsn{namespace="{{ .namespace }}",pod=~"{{ .podSelector }}"} + - cnpg_pg_stat_subscription_latest_end_lsn{namespace="{{ .namespace }}",pod=~"{{ .podSelector }}"} + ) / 1024^3 > 4 +for: 2m +labels: + severity: critical + namespace: {{ .namespace }} + cnpg_cluster: {{ .cluster }} + alert_type: logical-replication +{{- range $key, $val := .additionalLabels }} + {{ $key }}: {{ $val | toString | quote }} +{{- end }} +{{- end -}} diff --git a/charts/paradedb/prometheus_rules/cluster-logical_replication_distance-warning.yaml b/charts/paradedb/prometheus_rules/cluster-logical_replication_distance-warning.yaml new file mode 100644 index 0000000000..004b385244 --- /dev/null +++ b/charts/paradedb/prometheus_rules/cluster-logical_replication_distance-warning.yaml @@ -0,0 +1,25 @@ +{{- $alert := "CNPGClusterLogicalReplicationDistanceHigh" -}} +{{- if not (has $alert .excludeRules) -}} +alert: {{ $alert }} +annotations: + summary: ParadeDB logical replication WAL-position distance is warning + description: |- + ParadeDB CNPG Cluster "{{ .namespace }}/{{ .cluster }}" reports a WAL-position difference of {{ .value }} GiB as the logical replication subscriber for subscription "{{ .labels.subname }}" in database "{{ .labels.datname }}". + + This does not measure unapplied data. Check replication errors, worker health, and subscriber logs. + runbook_url: https://github.com/paradedb/charts/blob/main/charts/paradedb/docs/runbooks/CNPGClusterLogicalReplicationDistance.md +expr: | + max by (namespace, job, pod, datname, subname) ( + cnpg_pg_stat_subscription_received_lsn{namespace="{{ .namespace }}",pod=~"{{ .podSelector }}"} + - cnpg_pg_stat_subscription_latest_end_lsn{namespace="{{ .namespace }}",pod=~"{{ .podSelector }}"} + ) / 1024^3 > 1 +for: 5m +labels: + severity: warning + namespace: {{ .namespace }} + cnpg_cluster: {{ .cluster }} + alert_type: logical-replication +{{- range $key, $val := .additionalLabels }} + {{ $key }}: {{ $val | toString | quote }} +{{- end }} +{{- end -}} diff --git a/charts/paradedb/test/prometheus-rule-rendering/chainsaw-test.yaml b/charts/paradedb/test/prometheus-rule-rendering/chainsaw-test.yaml index bbf1358283..92d8832b69 100644 --- a/charts/paradedb/test/prometheus-rule-rendering/chainsaw-test.yaml +++ b/charts/paradedb/test/prometheus-rule-rendering/chainsaw-test.yaml @@ -78,6 +78,26 @@ spec: exit 1 fi + for alert in CNPGClusterLogicalReplicationDistanceHigh CNPGClusterLogicalReplicationDistanceCritical; do + distance_rule="$(printf '%s\n' "$rendered" | alert_rule "$alert")" + printf '%s\n' "$distance_rule" | grep -Fq 'cnpg_pg_stat_subscription_received_lsn' + printf '%s\n' "$distance_rule" | grep -Fq 'cnpg_pg_stat_subscription_latest_end_lsn' + printf '%s\n' "$distance_rule" | grep -Fq 'datname, subname' + if printf '%s\n' "$distance_rule" | grep -Fq 'kube_customresource'; then + echo "logical distance alert must not require custom-resource metrics" + exit 1 + fi + distance_excluded="$(render 3 --set "cluster.monitoring.prometheusRule.excludeRules[0]=$alert")" + if printf '%s\n' "$distance_excluded" | grep -Fq -- "- alert: $alert"; then + echo "excluded logical distance alert must not render" + exit 1 + fi + done + printf '%s\n' "$rendered" | alert_rule CNPGClusterLogicalReplicationDistanceHigh | grep -Fq '/ 1024^3 > 1' + printf '%s\n' "$rendered" | alert_rule CNPGClusterLogicalReplicationDistanceHigh | grep -Fq 'for: 5m' + printf '%s\n' "$rendered" | alert_rule CNPGClusterLogicalReplicationDistanceCritical | grep -Fq '/ 1024^3 > 4' + printf '%s\n' "$rendered" | alert_rule CNPGClusterLogicalReplicationDistanceCritical | grep -Fq 'for: 2m' + one_instance="$(render 1)" if printf '%s\n' "$one_instance" | grep -Fq 'CNPGClusterHA'; then echo "single-instance clusters must not render HA alerts"