From e56f4d4c3aaf2099effe421ba15e76cc4a83caa4 Mon Sep 17 00:00:00 2001 From: tornquist Date: Fri, 25 Sep 2026 18:03:39 +0000 Subject: [PATCH 1/9] Add typed deletion drain operator CronJob Co-authored-by: Codex Signed-off-by: tornquist --- ARCHITECTURE.md | 9 + deploy/charts/buzz/Chart.yaml | 4 +- deploy/charts/buzz/README.md | 22 +- .../charts/buzz/templates/_operator-jobs.tpl | 86 +++++++ deploy/charts/buzz/templates/_validate.tpl | 7 + .../templates/deletion-drain-cronjob.yaml | 4 + .../buzz/tests/deletion_drain_test.yaml | 214 ++++++++++++++++++ deploy/charts/buzz/values.schema.json | 28 +++ deploy/charts/buzz/values.yaml | 23 ++ docs/operator-community-deletion.md | 87 +++++++ 10 files changed, 481 insertions(+), 3 deletions(-) create mode 100644 deploy/charts/buzz/templates/_operator-jobs.tpl create mode 100644 deploy/charts/buzz/templates/deletion-drain-cronjob.yaml create mode 100644 deploy/charts/buzz/tests/deletion_drain_test.yaml create mode 100644 docs/operator-community-deletion.md diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 8aecbd61cc4..d30b9c9a2f5 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -750,10 +750,19 @@ Subcommands: | `remove-member` | Remove a pubkey from the relay membership list (`--pubkey`, optional `--role` guard); publishes kind:13534 roster | | `list-members` | List all relay members | | `generate-key` | Generate a new Nostr keypair (for bootstrapping) | +| `deletions` | Submit, inspect, approve, abort, unblock, run, or drain durable whole-community deletion requests | +| `storage-snapshot` | Run one isolated S3 accounting scan and publish its complete Postgres snapshot | | `reconcile-channels` | Emit kind:39000/39002 discovery events for channels missing them (idempotent) | The `buzz-admin` binary is shipped in the relay Docker image (`/usr/local/bin/buzz-admin`) and is the recommended way to manage relay membership in production. Use `./run.sh add-member`, `./run.sh remove-member`, and `./run.sh list-members` in Docker Compose deployments. +Kubernetes deployments may schedule the typed one-shot +`buzz-admin deletions drain` command directly. The pod owns its bounded +Postgres/Redis clients and S3 client; it does not call relay HTTP. Durable +requests, leases, retry timing, and checkpoints in Postgres are the handoff and +execution authority, so Kubernetes uses `Forbid` concurrency and zero Job +retries rather than introducing a second retry system. + --- ### buzz-test-client — Integration Test Harness diff --git a/deploy/charts/buzz/Chart.yaml b/deploy/charts/buzz/Chart.yaml index e26a2815fff..55d9db5488f 100644 --- a/deploy/charts/buzz/Chart.yaml +++ b/deploy/charts/buzz/Chart.yaml @@ -7,7 +7,7 @@ description: | PostgreSQL and Redis. Configurable for single-node evaluation (subcharts on) and HA production (external services, existingSecret). type: application -version: 0.1.9 +version: 0.1.10 appVersion: "0.1.0" home: https://github.com/block/buzz sources: @@ -24,7 +24,7 @@ maintainers: annotations: artifacthub.io/changes: | - kind: added - description: Optional isolated storage-accounting CronJob with durable relay snapshots. + description: Optional typed deletion-drain operator CronJob using durable database leases. artifacthub.io/license: Apache-2.0 # Optional eval-only subcharts. Production deploys disable both and point diff --git a/deploy/charts/buzz/README.md b/deploy/charts/buzz/README.md index f5075fd19d5..1bdff7592e9 100644 --- a/deploy/charts/buzz/README.md +++ b/deploy/charts/buzz/README.md @@ -12,7 +12,7 @@ This chart has two operating profiles selected by values: ## Quickstart (eval only) ```sh -helm install buzz oci://ghcr.io/block/buzz/charts/buzz --version 0.1.8 \ +helm install buzz oci://ghcr.io/block/buzz/charts/buzz --version 0.1.10 \ --create-namespace --namespace buzz \ --set quickstart=true \ --set postgresql.enabled=true \ @@ -116,6 +116,26 @@ is still parsed strictly, but reachability and addressing errors surface on the first storage operation. `/_readiness` tests no external dependency in either case — see the readiness contract below. +## Community deletion operator job + +`operatorJobs.deletionDrain` is a disabled-by-default, typed CronJob for +`/usr/local/bin/buzz-admin deletions drain`. It runs inside the relay image with +bounded Job lifetime/history, `concurrencyPolicy: Forbid`, `backoffLimit: 0`, +and no relay HTTP call. Postgres deletion requests, leases, retries, and +checkpoints remain the execution authority. + +The pod receives only `DATABASE_URL`, `REDIS_URL`, and required S3 +configuration/credential variables. It does not receive the relay private key, +git-hook secret, relay URL, service-account token, service links, or a generic +environment registry. Schedule, +deadline, history, termination grace, resources, service account, pod labels, +and pod annotations are independently configurable under +`operatorJobs.deletionDrain`. + +See [`docs/operator-community-deletion.md`](../../../docs/operator-community-deletion.md) +for enablement, permissions, the staffed first-run procedure, failure recovery, +and the current explicit approval/alerting boundaries. + ### Early-startup telemetry contract `buzz_process_lifecycle` JSON records are the authoritative history for the diff --git a/deploy/charts/buzz/templates/_operator-jobs.tpl b/deploy/charts/buzz/templates/_operator-jobs.tpl new file mode 100644 index 00000000000..aec91d52ff2 --- /dev/null +++ b/deploy/charts/buzz/templates/_operator-jobs.tpl @@ -0,0 +1,86 @@ +{{/* Closed rendering foundation for typed Buzz operator jobs. */}} + +{{- define "buzz.operatorCronJob" -}} +{{- $root := .root -}} +{{- if eq .type "deletionDrain" -}} +{{- $job := $root.Values.operatorJobs.deletionDrain -}} +apiVersion: batch/v1 +kind: CronJob +metadata: + name: {{ include "buzz.fullname" $root }}-deletion-drain + labels: + {{- include "buzz.labels" $root | nindent 4 }} + app.kubernetes.io/component: deletion-drain +spec: + schedule: {{ $job.schedule | quote }} + concurrencyPolicy: Forbid + successfulJobsHistoryLimit: {{ $job.successfulJobsHistoryLimit }} + failedJobsHistoryLimit: {{ $job.failedJobsHistoryLimit }} + jobTemplate: + spec: + activeDeadlineSeconds: {{ $job.activeDeadlineSeconds }} + backoffLimit: 0 + template: + metadata: + labels: + {{- include "buzz.selectorLabels" $root | nindent 12 }} + app.kubernetes.io/component: deletion-drain + {{- with $job.podLabels }} + {{- toYaml . | nindent 12 }} + {{- end }} + annotations: + {{- toYaml $job.podAnnotations | nindent 12 }} + spec: + restartPolicy: Never + terminationGracePeriodSeconds: {{ $job.terminationGracePeriodSeconds }} + serviceAccountName: {{ default (include "buzz.serviceAccountName" $root) $job.serviceAccountName }} + automountServiceAccountToken: false + enableServiceLinks: false + securityContext: + {{- toYaml $root.Values.relay.securityContext | nindent 12 }} + {{- with $root.Values.image.pullSecrets }} + imagePullSecrets: + {{- toYaml . | nindent 12 }} + {{- end }} + containers: + - name: deletion-drain + image: {{ include "buzz.image" $root }} + imagePullPolicy: {{ $root.Values.image.pullPolicy }} + securityContext: + {{- omit $root.Values.relay.containerSecurityContext "readOnlyRootFilesystem" | toYaml | nindent 16 }} + readOnlyRootFilesystem: true + command: ["/usr/local/bin/buzz-admin"] + args: ["deletions", "drain"] + env: + - { name: BUZZ_S3_ENDPOINT, value: {{ required "s3.endpoint is required when operatorJobs.deletionDrain.enabled=true" (include "buzz.s3Endpoint" $root) | quote }} } + - { name: BUZZ_S3_BUCKET, value: {{ required "s3.bucket is required when operatorJobs.deletionDrain.enabled=true" $root.Values.s3.bucket | quote }} } + - { name: BUZZ_S3_REGION, value: {{ $root.Values.s3.region | quote }} } + - { name: BUZZ_S3_ADDRESSING_STYLE, value: {{ $root.Values.s3.addressingStyle | quote }} } + - name: DATABASE_URL + valueFrom: + secretKeyRef: + name: {{ include "buzz.envSecretName" $root }} + key: DATABASE_URL + - name: REDIS_URL + valueFrom: + secretKeyRef: + name: {{ include "buzz.envSecretName" $root }} + key: REDIS_URL + - name: BUZZ_S3_ACCESS_KEY + valueFrom: + secretKeyRef: + name: {{ include "buzz.envSecretName" $root }} + key: BUZZ_S3_ACCESS_KEY + optional: true + - name: BUZZ_S3_SECRET_KEY + valueFrom: + secretKeyRef: + name: {{ include "buzz.envSecretName" $root }} + key: BUZZ_S3_SECRET_KEY + optional: true + resources: + {{- toYaml $job.resources | nindent 16 }} +{{- else -}} +{{- fail (printf "unsupported typed operator job %q" .type) -}} +{{- end -}} +{{- end -}} diff --git a/deploy/charts/buzz/templates/_validate.tpl b/deploy/charts/buzz/templates/_validate.tpl index aa7f7ac13cf..8d055f825ed 100644 --- a/deploy/charts/buzz/templates/_validate.tpl +++ b/deploy/charts/buzz/templates/_validate.tpl @@ -18,6 +18,13 @@ surface at template time regardless of which manifest helm renders first. {{- end -}} {{- end -}} +{{/* The deletion executor always uses Redis for tenant-scoped invalidation. */}} +{{- if .Values.operatorJobs.deletionDrain.enabled -}} + {{- if and (not .Values.redis.enabled) (not .Values.externalRedis.url) (not .Values.secrets.existingSecret) -}} + {{- fail "operatorJobs.deletionDrain requires Redis. Enable redis.enabled=true, set externalRedis.url, or provide secrets.existingSecret with key REDIS_URL." -}} + {{- end -}} +{{- end -}} + {{/* Multiple replicas do NOT require ReadWriteMany git storage. Git ref/object state is object-store-backed: every read and write hydrates diff --git a/deploy/charts/buzz/templates/deletion-drain-cronjob.yaml b/deploy/charts/buzz/templates/deletion-drain-cronjob.yaml new file mode 100644 index 00000000000..84acc56590a --- /dev/null +++ b/deploy/charts/buzz/templates/deletion-drain-cronjob.yaml @@ -0,0 +1,4 @@ +{{- include "buzz.validate" . -}} +{{- if .Values.operatorJobs.deletionDrain.enabled }} +{{- include "buzz.operatorCronJob" (dict "root" . "type" "deletionDrain") }} +{{- end }} diff --git a/deploy/charts/buzz/tests/deletion_drain_test.yaml b/deploy/charts/buzz/tests/deletion_drain_test.yaml new file mode 100644 index 00000000000..18e0d1a9ef2 --- /dev/null +++ b/deploy/charts/buzz/tests/deletion_drain_test.yaml @@ -0,0 +1,214 @@ +suite: typed deletion drain operator job +templates: + - templates/deletion-drain-cronjob.yaml +tests: + - it: renders no deletion executor by default + set: + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + externalPostgresql.url: postgres://u:p@h:5432/d + externalRedis.url: redis://h:6379 + s3.endpoint: https://s3.example.com + s3.accessKey: test + s3.secretKey: test + asserts: + - hasDocuments: + count: 0 + + - it: renders the bounded typed drain command with isolated pod controls + set: + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + externalPostgresql.url: postgres://u:p@h:5432/d + externalRedis.url: redis://h:6379 + s3.endpoint: https://s3.example.com + s3.bucket: buzz-media-example + s3.region: us-west-2 + s3.addressingStyle: virtual + s3.accessKey: test + s3.secretKey: test + operatorJobs.deletionDrain: + enabled: true + schedule: "*/7 * * * *" + activeDeadlineSeconds: 900 + successfulJobsHistoryLimit: 2 + failedJobsHistoryLimit: 4 + terminationGracePeriodSeconds: 45 + serviceAccountName: buzz-deletion-drain + podLabels: + tags.datadoghq.com/service: buzz-deletion-drain + podAnnotations: + sidecar.istio.io/inject: "false" + example.com/operator-job: deletion-drain + resources: + requests: + cpu: 250m + memory: 256Mi + limits: + cpu: "1" + memory: 1Gi + asserts: + - hasDocuments: + count: 1 + - isAPIVersion: + of: batch/v1 + - isKind: + of: CronJob + - equal: + path: metadata.name + value: RELEASE-NAME-buzz-deletion-drain + - equal: + path: metadata.labels["app.kubernetes.io/component"] + value: deletion-drain + - equal: + path: spec.schedule + value: "*/7 * * * *" + - equal: + path: spec.concurrencyPolicy + value: Forbid + - equal: + path: spec.successfulJobsHistoryLimit + value: 2 + - equal: + path: spec.failedJobsHistoryLimit + value: 4 + - equal: + path: spec.jobTemplate.spec.activeDeadlineSeconds + value: 900 + - equal: + path: spec.jobTemplate.spec.backoffLimit + value: 0 + - equal: + path: spec.jobTemplate.spec.template.spec.restartPolicy + value: Never + - equal: + path: spec.jobTemplate.spec.template.spec.terminationGracePeriodSeconds + value: 45 + - equal: + path: spec.jobTemplate.spec.template.spec.serviceAccountName + value: buzz-deletion-drain + - equal: + path: spec.jobTemplate.spec.template.spec.automountServiceAccountToken + value: false + - equal: + path: spec.jobTemplate.spec.template.spec.enableServiceLinks + value: false + - equal: + path: spec.jobTemplate.spec.template.metadata.labels["tags.datadoghq.com/service"] + value: buzz-deletion-drain + - equal: + path: spec.jobTemplate.spec.template.metadata.annotations["example.com/operator-job"] + value: deletion-drain + - equal: + path: spec.jobTemplate.spec.template.spec.containers[0].securityContext.readOnlyRootFilesystem + value: true + - equal: + path: spec.jobTemplate.spec.template.spec.containers[0].command + value: ["/usr/local/bin/buzz-admin"] + - equal: + path: spec.jobTemplate.spec.template.spec.containers[0].args + value: ["deletions", "drain"] + - equal: + path: spec.jobTemplate.spec.template.spec.containers[0].resources.requests.cpu + value: 250m + - equal: + path: spec.jobTemplate.spec.template.spec.containers[0].resources.limits.memory + value: 1Gi + + - it: exposes only database redis and deletion S3 configuration + set: + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + externalPostgresql.url: postgres://u:p@h:5432/d + externalRedis.url: redis://h:6379 + secrets.existingSecret: buzz-operator-secrets + s3.endpoint: https://s3.example.com + s3.bucket: buzz-media-example + s3.region: us-west-2 + s3.addressingStyle: virtual + operatorJobs.deletionDrain.enabled: true + asserts: + - lengthEqual: + path: spec.jobTemplate.spec.template.spec.containers[0].env + count: 8 + - notExists: + path: spec.jobTemplate.spec.template.spec.containers[0].envFrom + - contains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: DATABASE_URL + valueFrom: + secretKeyRef: + name: buzz-operator-secrets + key: DATABASE_URL + - contains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: REDIS_URL + valueFrom: + secretKeyRef: + name: buzz-operator-secrets + key: REDIS_URL + - contains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: BUZZ_S3_ENDPOINT + value: https://s3.example.com + - contains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: BUZZ_S3_BUCKET + value: buzz-media-example + - contains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: BUZZ_S3_REGION + value: us-west-2 + - contains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: BUZZ_S3_ADDRESSING_STYLE + value: virtual + - contains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: BUZZ_S3_ACCESS_KEY + valueFrom: + secretKeyRef: + name: buzz-operator-secrets + key: BUZZ_S3_ACCESS_KEY + optional: true + - contains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: BUZZ_S3_SECRET_KEY + valueFrom: + secretKeyRef: + name: buzz-operator-secrets + key: BUZZ_S3_SECRET_KEY + optional: true + - notContains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: BUZZ_RELAY_PRIVATE_KEY + - notContains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: BUZZ_GIT_HOOK_HMAC_SECRET + - notContains: + path: spec.jobTemplate.spec.template.spec.containers[0].env + content: + name: RELAY_URL + + - it: rejects enablement without a Redis source + set: + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + externalPostgresql.url: postgres://u:p@h:5432/d + s3.endpoint: https://s3.example.com + s3.accessKey: test + s3.secretKey: test + operatorJobs.deletionDrain.enabled: true + asserts: + - failedTemplate: + errorPattern: "operatorJobs.deletionDrain requires Redis" diff --git a/deploy/charts/buzz/values.schema.json b/deploy/charts/buzz/values.schema.json index ea5db6398dd..309b2f5c37f 100644 --- a/deploy/charts/buzz/values.schema.json +++ b/deploy/charts/buzz/values.schema.json @@ -234,6 +234,34 @@ "resources": { "type": "object" } } }, + "operatorJobs": { + "type": "object", + "additionalProperties": false, + "properties": { + "deletionDrain": { + "type": "object", + "additionalProperties": false, + "properties": { + "enabled": { "type": "boolean" }, + "schedule": { "type": "string", "minLength": 1 }, + "activeDeadlineSeconds": { "type": "integer", "minimum": 1, "maximum": 86400 }, + "successfulJobsHistoryLimit": { "type": "integer", "minimum": 0, "maximum": 10 }, + "failedJobsHistoryLimit": { "type": "integer", "minimum": 0, "maximum": 10 }, + "terminationGracePeriodSeconds": { "type": "integer", "minimum": 1, "maximum": 300 }, + "serviceAccountName": { "type": "string" }, + "podLabels": { + "type": "object", + "additionalProperties": { "type": "string" } + }, + "podAnnotations": { + "type": "object", + "additionalProperties": { "type": "string" } + }, + "resources": { "type": "object" } + } + } + } + }, "minio": { "type": "object", "additionalProperties": false, diff --git a/deploy/charts/buzz/values.yaml b/deploy/charts/buzz/values.yaml index 496c32e979f..f43925346e5 100644 --- a/deploy/charts/buzz/values.yaml +++ b/deploy/charts/buzz/values.yaml @@ -391,6 +391,29 @@ storageAccounting: cpu: "1" memory: 20Gi +# Typed, one-shot operator jobs. The chart intentionally exposes no generic +# command or environment registry: each job has a reviewed executable and +# least-privilege environment contract. +operatorJobs: + deletionDrain: + enabled: false + schedule: "*/5 * * * *" + activeDeadlineSeconds: 3600 + successfulJobsHistoryLimit: 1 + failedJobsHistoryLimit: 3 + terminationGracePeriodSeconds: 30 + serviceAccountName: "" # defaults to the main Buzz service account + podLabels: {} + podAnnotations: + sidecar.istio.io/inject: "false" + resources: + requests: + cpu: 100m + memory: 256Mi + limits: + cpu: "1" + memory: 1Gi + # In-cluster MinIO for the quickstart profile only. Production deploys leave # this disabled and use s3.* (or secrets.existingSecret) against managed S3. minio: diff --git a/docs/operator-community-deletion.md b/docs/operator-community-deletion.md new file mode 100644 index 00000000000..ec2c20f0004 --- /dev/null +++ b/docs/operator-community-deletion.md @@ -0,0 +1,87 @@ +# Community Deletion Operator Job + +Buzz executes whole-community deletion through the typed, one-shot +`/usr/local/bin/buzz-admin deletions drain` command. The Helm chart can schedule +that command as a Kubernetes CronJob; it does not call relay HTTP and it does +not add another queue or retry service. + +Postgres remains the handoff and source of truth. A run claims only requests +that the deletion store considers runnable, heartbeats the existing lease, and +resumes from durable checkpoints. `concurrencyPolicy: Forbid` prevents scheduled +pod overlap, `backoffLimit: 0` prevents Kubernetes Job retries, and the deletion +store remains authoritative when a pod exits, reaches its deadline, or is +replaced. + +## Enablement + +The CronJob is disabled by default. Production deployments should use an +existing Secret and a dedicated service account when their cluster policy +supports one: + +```yaml +secrets: + existingSecret: buzz-operator-secrets + +operatorJobs: + deletionDrain: + enabled: true + schedule: "*/5 * * * *" + activeDeadlineSeconds: 3600 + terminationGracePeriodSeconds: 30 + serviceAccountName: buzz-deletion-drain + podLabels: + tags.datadoghq.com/service: buzz-deletion-drain + podAnnotations: + sidecar.istio.io/inject: "false" + resources: + requests: + cpu: 100m + memory: 256Mi + limits: + cpu: "1" + memory: 1Gi +``` + +Set `s3.endpoint`, `s3.bucket`, `s3.region`, and `s3.addressingStyle` in chart +values. The selected Secret must contain `DATABASE_URL` and `REDIS_URL`; it may +contain `BUZZ_S3_ACCESS_KEY` and `BUZZ_S3_SECRET_KEY` when the object store uses +static credentials. The pod receives only those connection values and the four +non-secret S3 settings. It does not receive `BUZZ_RELAY_PRIVATE_KEY`, +`BUZZ_GIT_HOOK_HMAC_SECRET`, `RELAY_URL`, or the full Secret through `envFrom`. +The pod also disables service-account token automounting and Kubernetes service +link environment injection because the executor does not call the Kubernetes +API or discover cluster Services. + +The S3 principal needs the relay's normal object permissions plus bucket-level +`s3:ListBucketVersions` and object-level `s3:DeleteObjectVersion` for every +tenant-owned prefix. This also applies to never-versioned buckets because S3 +reports their objects with the `null` version id. + +## Runbook + +1. Confirm database migrations are current and the deletion request has crossed + the explicit inventory and approval boundary with + `buzz-admin deletions inspect `. +2. Confirm the selected Secret contains the required keys and the S3 principal + has version-list and exact-version delete permissions. +3. Enable the CronJob and inspect its rendered command and environment before + rollout. +4. Start one staffed manual run with + `kubectl create job --from=cronjob/-buzz-deletion-drain `. +5. Follow pod logs and re-run `buzz-admin deletions inspect ` to + verify lease, checkpoint, retry, blocked, and terminal state. +6. If a run fails or times out, fix the recorded dependency or permission + failure. Do not add Kubernetes retries: the next scheduled drain consults the + durable retry/checkpoint state and resumes only when the store allows it. + +The current owner self-serve relay admission records an owner-origin request at +`submitted` and intentionally performs no inventory or approval synchronously. +The deletion engine rejects `submitted` and `inventoried` requests at its +explicit approval boundary. Automating the privileged inventory/approval step +is therefore a separate control-plane slice; enabling this CronJob alone does +not make a newly accepted owner request destructive. + +The chart has no existing PrometheusRule or provider-neutral CronJob alert +integration. Operators must alert on failed/missed Jobs and long-running active +Jobs in their deployment platform. Adding a chart-native alert abstraction is +debt, not part of this job contract. From 07312054dbff1618485be2af333b1a94f57be5dc Mon Sep 17 00:00:00 2001 From: tornquist Date: Fri, 25 Sep 2026 18:03:39 +0000 Subject: [PATCH 2/9] Reject reserved deletion job pod labels Co-authored-by: Codex Signed-off-by: tornquist --- .../charts/buzz/templates/_operator-jobs.tpl | 5 +++++ .../charts/buzz/tests/deletion_drain_test.yaml | 18 ++++++++++++++++++ 2 files changed, 23 insertions(+) diff --git a/deploy/charts/buzz/templates/_operator-jobs.tpl b/deploy/charts/buzz/templates/_operator-jobs.tpl index aec91d52ff2..f273fe8ebd3 100644 --- a/deploy/charts/buzz/templates/_operator-jobs.tpl +++ b/deploy/charts/buzz/templates/_operator-jobs.tpl @@ -4,6 +4,11 @@ {{- $root := .root -}} {{- if eq .type "deletionDrain" -}} {{- $job := $root.Values.operatorJobs.deletionDrain -}} +{{- range $label := list "app.kubernetes.io/name" "app.kubernetes.io/instance" "app.kubernetes.io/component" -}} +{{- if hasKey $job.podLabels $label -}} +{{- fail (printf "operatorJobs.deletionDrain.podLabels may not set chart-owned label %q" $label) -}} +{{- end -}} +{{- end -}} apiVersion: batch/v1 kind: CronJob metadata: diff --git a/deploy/charts/buzz/tests/deletion_drain_test.yaml b/deploy/charts/buzz/tests/deletion_drain_test.yaml index 18e0d1a9ef2..5053c9911d3 100644 --- a/deploy/charts/buzz/tests/deletion_drain_test.yaml +++ b/deploy/charts/buzz/tests/deletion_drain_test.yaml @@ -200,6 +200,24 @@ tests: content: name: RELAY_URL + - it: rejects chart-owned pod identity labels + set: + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + externalPostgresql.url: postgres://u:p@h:5432/d + externalRedis.url: redis://h:6379 + s3.endpoint: https://s3.example.com + s3.bucket: buzz-media-example + s3.accessKey: test + s3.secretKey: test + operatorJobs.deletionDrain: + enabled: true + podLabels: + app.kubernetes.io/component: relay + asserts: + - failedTemplate: + errorPattern: 'operatorJobs.deletionDrain.podLabels may not set chart-owned label "app.kubernetes.io/component"' + - it: rejects enablement without a Redis source set: relayUrl: wss://buzz.example.com From 5c3af3b4c012da1166a2ef1e440d1c9e6892ca21 Mon Sep 17 00:00:00 2001 From: tornquist Date: Fri, 25 Sep 2026 18:03:39 +0000 Subject: [PATCH 3/9] Correct the deletion drain runbook's operational claims MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The runbook told operators to target `cronjob/-buzz-deletion-drain`. `buzz.fullname` collapses to the release name when it already contains the chart name, so the documented name is wrong for the documented install: `helm install buzz ...` renders `buzz-deletion-drain`. Discover the CronJob by its component label instead, and describe the name rule rather than a single guessed spelling. Three more corrections: `activeDeadlineSeconds` was presented as if a timed-out run were just another retry. It is not recorded as one — shutdown releases the claim without recording a retry, and only the object-store drain resumes mid-stage — so a deadline landing repeatedly inside a non-resumable stage loops forever with a rising `attempts`, a flat `retry_count`, and no block. Document how to spot that from both the request and Kubernetes, how to size the deadline, and how to recover. `terminationGracePeriodSeconds` was presented as a clean handoff. Document that a pod still working at the end of the window is SIGKILLed holding its lease, and that recovery is lease expiry plus reclaim under a new generation. An empty `serviceAccountName` was described as a neutral default. It inherits the relay's service account; `automountServiceAccountToken: false` hides the projected token but does not detach cloud IAM bindings resolved through the node metadata path. Recommend a dedicated pre-created account when the executor's IAM blast radius should be smaller than the relay's. Co-Authored-By: Claude Opus 5 Signed-off-by: tornquist --- deploy/charts/buzz/values.yaml | 4 +- docs/operator-community-deletion.md | 67 ++++++++++++++++++++++++++--- 2 files changed, 64 insertions(+), 7 deletions(-) diff --git a/deploy/charts/buzz/values.yaml b/deploy/charts/buzz/values.yaml index f43925346e5..d51d37b4bb7 100644 --- a/deploy/charts/buzz/values.yaml +++ b/deploy/charts/buzz/values.yaml @@ -402,7 +402,9 @@ operatorJobs: successfulJobsHistoryLimit: 1 failedJobsHistoryLimit: 3 terminationGracePeriodSeconds: 30 - serviceAccountName: "" # defaults to the main Buzz service account + # Empty inherits the relay service account, and with it any cloud IAM + # binding on that account. Name a dedicated account to narrow the executor. + serviceAccountName: "" podLabels: {} podAnnotations: sidecar.istio.io/inject: "false" diff --git a/docs/operator-community-deletion.md b/docs/operator-community-deletion.md index ec2c20f0004..d7fd69b3a2a 100644 --- a/docs/operator-community-deletion.md +++ b/docs/operator-community-deletion.md @@ -15,8 +15,7 @@ replaced. ## Enablement The CronJob is disabled by default. Production deployments should use an -existing Secret and a dedicated service account when their cluster policy -supports one: +existing Secret and a dedicated, pre-created service account: ```yaml secrets: @@ -52,6 +51,16 @@ The pod also disables service-account token automounting and Kubernetes service link environment injection because the executor does not call the Kubernetes API or discover cluster Services. +Leaving `operatorJobs.deletionDrain.serviceAccountName` empty falls back to the +relay's own service account. `automountServiceAccountToken: false` only +suppresses the projected token inside the pod; it does not detach the identity. +Cloud IAM bindings attached to that service account — IRSA on EKS, Workload +Identity on GKE — are resolved by the node/metadata path and still apply, so the +drain pod inherits the relay's cloud permissions. Create a dedicated service +account with only the object-store permissions listed below and name it +explicitly if you want the executor's IAM blast radius to be smaller than the +relay's. + The S3 principal needs the relay's normal object permissions plus bucket-level `s3:ListBucketVersions` and object-level `s3:DeleteObjectVersion` for every tenant-owned prefix. This also applies to never-versioned buckets because S3 @@ -66,14 +75,60 @@ reports their objects with the `null` version id. has version-list and exact-version delete permissions. 3. Enable the CronJob and inspect its rendered command and environment before rollout. -4. Start one staffed manual run with - `kubectl create job --from=cronjob/-buzz-deletion-drain `. -5. Follow pod logs and re-run `buzz-admin deletions inspect ` to +4. Locate the rendered CronJob by label rather than by guessing its name: + + ```sh + kubectl get cronjob -n \ + -l app.kubernetes.io/component=deletion-drain,app.kubernetes.io/instance= + ``` + + The name is `-deletion-drain`. `buzz.fullname` collapses to the + release name when the release name already contains the chart name, so + `helm install buzz ...` renders `buzz-deletion-drain`, not + `buzz-buzz-deletion-drain`. +5. Start one staffed manual run with + `kubectl create job --from=cronjob/ `. +6. Follow pod logs and re-run `buzz-admin deletions inspect ` to verify lease, checkpoint, retry, blocked, and terminal state. -6. If a run fails or times out, fix the recorded dependency or permission +7. If a run fails or times out, fix the recorded dependency or permission failure. Do not add Kubernetes retries: the next scheduled drain consults the durable retry/checkpoint state and resumes only when the store allows it. +## Deadlines, termination, and the retry budget + +`activeDeadlineSeconds` is a Kubernetes-side limit, and the deletion store does +not learn why a pod went away. Two consequences matter when reading state: + +- A `DeadlineExceeded` Job is not recorded as a deletion retry. Shutdown + releases the claim without recording one, so `retry_count` and `blocked_at` + do not advance. Only the object-store drain checkpoints per manifest chunk + and resumes mid-stage; every other stage restarts from its beginning on the + next run. A deadline that keeps landing inside one of those non-resumable + stages therefore repeats indefinitely: each run increments `attempts` and + burns the window again while the retry budget never moves and the request is + never blocked. +- Diagnose this from both sides. `buzz-admin deletions inspect ` + shows a rising `attempts` with a flat `retry_count` and no `last_error`; + Kubernetes holds the reason. The chart labels the CronJob and the drain pods, + but not the generated Jobs, so find the attempts with + `kubectl get pods -n -l app.kubernetes.io/component=deletion-drain` + and read the condition with `kubectl describe job `. + +Size `activeDeadlineSeconds` for the longest single stage this community will +run, not for the average run. To recover, either raise the deadline and let the +schedule pick the request back up, or take one staffed run with +`buzz-admin deletions run ` outside the CronJob's deadline. + +`terminationGracePeriodSeconds` is a best-effort window, not a guarantee. The +drain command handles `SIGTERM` and releases its lease cleanly when it wins the +race, but a pod that is still working when the grace period expires is +`SIGKILL`ed with the lease still held. Nothing is lost: the durable lease simply +expires (60s by default, heartbeated every 10s) and the next run reclaims the +request with a fresh lease generation, which fences any straggler write from the +killed process. Expect up to roughly a lease duration of delay before the +request is runnable again; do not raise the grace period expecting a clean +handoff. + The current owner self-serve relay admission records an owner-origin request at `submitted` and intentionally performs no inventory or approval synchronously. The deletion engine rejects `submitted` and `inventoried` requests at its From d791f86a3d342dd86a6656703bacfd817a9726ed Mon Sep 17 00:00:00 2001 From: tornquist Date: Mon, 28 Sep 2026 13:24:52 +0000 Subject: [PATCH 4/9] Cover deletion drain chart contracts Exercise both supported credential shapes through the CI render matrix and bind the typed job guards, schema boundary, pod identity, and default sidecar annotation. Co-authored-by: Codex Signed-off-by: tornquist --- .../buzz/tests/deletion_drain_test.yaml | 81 ++++++++++++++++++- .../deletion-drain-bundled-values.yaml | 11 +++ .../production-existing-secret-values.yaml | 3 + 3 files changed, 93 insertions(+), 2 deletions(-) create mode 100644 deploy/charts/buzz/tests/fixtures/deletion-drain-bundled-values.yaml diff --git a/deploy/charts/buzz/tests/deletion_drain_test.yaml b/deploy/charts/buzz/tests/deletion_drain_test.yaml index 5053c9911d3..1f8a6315ced 100644 --- a/deploy/charts/buzz/tests/deletion_drain_test.yaml +++ b/deploy/charts/buzz/tests/deletion_drain_test.yaml @@ -38,7 +38,6 @@ tests: podLabels: tags.datadoghq.com/service: buzz-deletion-drain podAnnotations: - sidecar.istio.io/inject: "false" example.com/operator-job: deletion-drain resources: requests: @@ -96,6 +95,18 @@ tests: - equal: path: spec.jobTemplate.spec.template.metadata.labels["tags.datadoghq.com/service"] value: buzz-deletion-drain + - equal: + path: spec.jobTemplate.spec.template.metadata.labels["app.kubernetes.io/name"] + value: buzz + - equal: + path: spec.jobTemplate.spec.template.metadata.labels["app.kubernetes.io/instance"] + value: RELEASE-NAME + - equal: + path: spec.jobTemplate.spec.template.metadata.labels["app.kubernetes.io/component"] + value: deletion-drain + - equal: + path: spec.jobTemplate.spec.template.metadata.annotations["sidecar.istio.io/inject"] + value: "false" - equal: path: spec.jobTemplate.spec.template.metadata.annotations["example.com/operator-job"] value: deletion-drain @@ -200,7 +211,43 @@ tests: content: name: RELAY_URL - - it: rejects chart-owned pod identity labels + - it: rejects the chart-owned name pod label with an explicit guard error + set: + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + externalPostgresql.url: postgres://u:p@h:5432/d + externalRedis.url: redis://h:6379 + s3.endpoint: https://s3.example.com + s3.bucket: buzz-media-example + s3.accessKey: test + s3.secretKey: test + operatorJobs.deletionDrain: + enabled: true + podLabels: + app.kubernetes.io/name: relay + asserts: + - failedTemplate: + errorPattern: 'operatorJobs.deletionDrain.podLabels may not set chart-owned label "app.kubernetes.io/name"' + + - it: rejects the chart-owned instance pod label with an explicit guard error + set: + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + externalPostgresql.url: postgres://u:p@h:5432/d + externalRedis.url: redis://h:6379 + s3.endpoint: https://s3.example.com + s3.bucket: buzz-media-example + s3.accessKey: test + s3.secretKey: test + operatorJobs.deletionDrain: + enabled: true + podLabels: + app.kubernetes.io/instance: another-release + asserts: + - failedTemplate: + errorPattern: 'operatorJobs.deletionDrain.podLabels may not set chart-owned label "app.kubernetes.io/instance"' + + - it: rejects the chart-owned component pod label with an explicit guard error set: relayUrl: wss://buzz.example.com ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" @@ -218,6 +265,36 @@ tests: - failedTemplate: errorPattern: 'operatorJobs.deletionDrain.podLabels may not set chart-owned label "app.kubernetes.io/component"' + - it: rejects existing-secret enablement without an explicit S3 endpoint + set: + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + secrets.existingSecret: buzz-operator-secrets + externalPostgresql.url: postgres://u:p@h:5432/d + externalRedis.url: redis://h:6379 + s3.bucket: buzz-media-example + operatorJobs.deletionDrain.enabled: true + asserts: + - failedTemplate: + errorPattern: "s3.endpoint is required when operatorJobs.deletionDrain.enabled=true" + + - it: rejects unknown deletion drain keys through the values schema + set: + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + externalPostgresql.url: postgres://u:p@h:5432/d + externalRedis.url: redis://h:6379 + s3.endpoint: https://s3.example.com + s3.bucket: buzz-media-example + s3.accessKey: test + s3.secretKey: test + operatorJobs.deletionDrain: + enabled: true + arbitraryCommand: ["unsafe"] + asserts: + - failedTemplate: + errorPattern: "Additional property arbitraryCommand is not allowed" + - it: rejects enablement without a Redis source set: relayUrl: wss://buzz.example.com diff --git a/deploy/charts/buzz/tests/fixtures/deletion-drain-bundled-values.yaml b/deploy/charts/buzz/tests/fixtures/deletion-drain-bundled-values.yaml new file mode 100644 index 00000000000..18d30a696da --- /dev/null +++ b/deploy/charts/buzz/tests/fixtures/deletion-drain-bundled-values.yaml @@ -0,0 +1,11 @@ +relayUrl: wss://buzz.example.com +ownerPubkey: "abcdef0123456789abcdef0123456789abcdef0123456789abcdef0123456789" +postgresql: + enabled: true +redis: + enabled: true +minio: + enabled: true +operatorJobs: + deletionDrain: + enabled: true diff --git a/deploy/charts/buzz/tests/fixtures/production-existing-secret-values.yaml b/deploy/charts/buzz/tests/fixtures/production-existing-secret-values.yaml index fd4bdadfeda..7020fb32b0b 100644 --- a/deploy/charts/buzz/tests/fixtures/production-existing-secret-values.yaml +++ b/deploy/charts/buzz/tests/fixtures/production-existing-secret-values.yaml @@ -8,6 +8,9 @@ externalPostgresql: url: "postgres://buzz:pw@postgres.example.com:5432/buzz" externalRedis: url: "redis://:pw@redis.example.com:6379" +operatorJobs: + deletionDrain: + enabled: true s3: endpoint: "https://s3.us-east-1.amazonaws.com" bucket: "buzz-media" From c0a392a48788d81ac3adfe663f5647aa426f1607 Mon Sep 17 00:00:00 2001 From: tornquist Date: Mon, 28 Sep 2026 13:25:36 +0000 Subject: [PATCH 5/9] Bound operator CronJob names Share a suffix-preserving 52-character name helper between the deletion drain and storage accounting CronJobs while retaining their existing short names. Co-authored-by: Codex Signed-off-by: tornquist --- deploy/charts/buzz/templates/_helpers.tpl | 7 ++++++ .../charts/buzz/templates/_operator-jobs.tpl | 2 +- .../templates/storage-accounting-cronjob.yaml | 2 +- .../buzz/tests/deletion_drain_test.yaml | 19 ++++++++++++++++ .../buzz/tests/storage_accounting_test.yaml | 22 +++++++++++++++++++ docs/operator-community-deletion.md | 6 +++-- 6 files changed, 54 insertions(+), 4 deletions(-) diff --git a/deploy/charts/buzz/templates/_helpers.tpl b/deploy/charts/buzz/templates/_helpers.tpl index e5ec9182d68..c00e643fdb3 100644 --- a/deploy/charts/buzz/templates/_helpers.tpl +++ b/deploy/charts/buzz/templates/_helpers.tpl @@ -17,6 +17,13 @@ {{- end -}} {{- end -}} +{{/* Kubernetes CronJob names are limited to 52 characters. */}} +{{- define "buzz.cronJobName" -}} +{{- $maxBaseLength := sub 51 (len .suffix) | int -}} +{{- $base := include "buzz.fullname" .root | trunc $maxBaseLength | trimSuffix "-" -}} +{{- printf "%s-%s" $base .suffix -}} +{{- end -}} + {{- define "buzz.chart" -}} {{- printf "%s-%s" .Chart.Name .Chart.Version | replace "+" "_" | trunc 63 | trimSuffix "-" -}} {{- end -}} diff --git a/deploy/charts/buzz/templates/_operator-jobs.tpl b/deploy/charts/buzz/templates/_operator-jobs.tpl index f273fe8ebd3..49a204f9ec9 100644 --- a/deploy/charts/buzz/templates/_operator-jobs.tpl +++ b/deploy/charts/buzz/templates/_operator-jobs.tpl @@ -12,7 +12,7 @@ apiVersion: batch/v1 kind: CronJob metadata: - name: {{ include "buzz.fullname" $root }}-deletion-drain + name: {{ include "buzz.cronJobName" (dict "root" $root "suffix" "deletion-drain") }} labels: {{- include "buzz.labels" $root | nindent 4 }} app.kubernetes.io/component: deletion-drain diff --git a/deploy/charts/buzz/templates/storage-accounting-cronjob.yaml b/deploy/charts/buzz/templates/storage-accounting-cronjob.yaml index deaa816888e..f753bd3b412 100644 --- a/deploy/charts/buzz/templates/storage-accounting-cronjob.yaml +++ b/deploy/charts/buzz/templates/storage-accounting-cronjob.yaml @@ -2,7 +2,7 @@ apiVersion: batch/v1 kind: CronJob metadata: - name: {{ include "buzz.fullname" . }}-storage-accounting + name: {{ include "buzz.cronJobName" (dict "root" . "suffix" "storage-accounting") }} labels: {{- include "buzz.labels" . | nindent 4 }} app.kubernetes.io/component: storage-accounting diff --git a/deploy/charts/buzz/tests/deletion_drain_test.yaml b/deploy/charts/buzz/tests/deletion_drain_test.yaml index 1f8a6315ced..cf2b7e8556a 100644 --- a/deploy/charts/buzz/tests/deletion_drain_test.yaml +++ b/deploy/charts/buzz/tests/deletion_drain_test.yaml @@ -307,3 +307,22 @@ tests: asserts: - failedTemplate: errorPattern: "operatorJobs.deletionDrain requires Redis" + + - it: preserves the deletion suffix while bounding a long release and name override + release: + name: very-long-release-name-for-operator-jobs + set: + nameOverride: custom-buzz-name + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + externalPostgresql.url: postgres://u:p@h:5432/d + externalRedis.url: redis://h:6379 + s3.endpoint: https://s3.example.com + s3.bucket: buzz-media-example + s3.accessKey: test + s3.secretKey: test + operatorJobs.deletionDrain.enabled: true + asserts: + - equal: + path: metadata.name + value: very-long-release-name-for-operator-j-deletion-drain diff --git a/deploy/charts/buzz/tests/storage_accounting_test.yaml b/deploy/charts/buzz/tests/storage_accounting_test.yaml index d37d8aefe98..563e1ec2283 100644 --- a/deploy/charts/buzz/tests/storage_accounting_test.yaml +++ b/deploy/charts/buzz/tests/storage_accounting_test.yaml @@ -53,6 +53,10 @@ tests: path: kind value: CronJob template: templates/storage-accounting-cronjob.yaml + - equal: + path: metadata.name + value: RELEASE-NAME-buzz-storage-accounting + template: templates/storage-accounting-cronjob.yaml - equal: path: spec.concurrencyPolicy value: Forbid @@ -143,3 +147,21 @@ tests: content: name: BUZZ_GIT_HOOK_HMAC_SECRET template: templates/storage-accounting-cronjob.yaml + + - it: preserves the storage suffix without a trailing separator after truncation + set: + fullnameOverride: storage-accounting-fullname-that-needs-truncation-at-the-separator + relayUrl: wss://buzz.example.com + ownerPubkey: "0000000000000000000000000000000000000000000000000000000000000000" + externalPostgresql.url: postgres://u:p@h:5432/d + externalRedis.url: redis://h:6379 + s3.endpoint: https://s3.example.com + s3.bucket: buzz-media-example + s3.accessKey: test + s3.secretKey: test + storageAccounting.enabled: true + asserts: + - equal: + path: metadata.name + value: storage-accounting-fullname-that-storage-accounting + template: templates/storage-accounting-cronjob.yaml diff --git a/docs/operator-community-deletion.md b/docs/operator-community-deletion.md index d7fd69b3a2a..546f73cf3d2 100644 --- a/docs/operator-community-deletion.md +++ b/docs/operator-community-deletion.md @@ -82,8 +82,10 @@ reports their objects with the `null` version id. -l app.kubernetes.io/component=deletion-drain,app.kubernetes.io/instance= ``` - The name is `-deletion-drain`. `buzz.fullname` collapses to the - release name when the release name already contains the chart name, so + The name is `-deletion-drain`. The chart truncates only the + fullname portion when needed so the suffix remains stable within Kubernetes' + 52-character CronJob name limit. `buzz.fullname` collapses to the release + name when the release name already contains the chart name, so `helm install buzz ...` renders `buzz-deletion-drain`, not `buzz-buzz-deletion-drain`. 5. Start one staffed manual run with From cdef0269269d36e51209cebd3d7d9e2c20ee4533 Mon Sep 17 00:00:00 2001 From: tornquist Date: Mon, 28 Sep 2026 13:25:59 +0000 Subject: [PATCH 6/9] Clarify deletion job workload identity Distinguish the disabled Kubernetes API token mount from provider workload-identity credentials while retaining the dedicated service-account recommendation. Co-authored-by: Codex Signed-off-by: tornquist --- deploy/charts/buzz/README.md | 6 ++++-- docs/operator-community-deletion.md | 20 +++++++++++--------- 2 files changed, 15 insertions(+), 11 deletions(-) diff --git a/deploy/charts/buzz/README.md b/deploy/charts/buzz/README.md index 1bdff7592e9..b290cff902d 100644 --- a/deploy/charts/buzz/README.md +++ b/deploy/charts/buzz/README.md @@ -126,8 +126,10 @@ checkpoints remain the execution authority. The pod receives only `DATABASE_URL`, `REDIS_URL`, and required S3 configuration/credential variables. It does not receive the relay private key, -git-hook secret, relay URL, service-account token, service links, or a generic -environment registry. Schedule, +git-hook secret, relay URL, service links, or a generic environment registry. +The chart disables the ordinary Kubernetes API service-account token mount; +platform workload-identity admission may still inject its own projected token +and provider environment variables. Schedule, deadline, history, termination grace, resources, service account, pod labels, and pod annotations are independently configurable under `operatorJobs.deletionDrain`. diff --git a/docs/operator-community-deletion.md b/docs/operator-community-deletion.md index 546f73cf3d2..61d8fd89456 100644 --- a/docs/operator-community-deletion.md +++ b/docs/operator-community-deletion.md @@ -49,17 +49,19 @@ non-secret S3 settings. It does not receive `BUZZ_RELAY_PRIVATE_KEY`, `BUZZ_GIT_HOOK_HMAC_SECRET`, `RELAY_URL`, or the full Secret through `envFrom`. The pod also disables service-account token automounting and Kubernetes service link environment injection because the executor does not call the Kubernetes -API or discover cluster Services. +API or discover cluster Services. This disables the ordinary Kubernetes API +token mount, not credentials injected by a platform workload-identity +mechanism. Leaving `operatorJobs.deletionDrain.serviceAccountName` empty falls back to the -relay's own service account. `automountServiceAccountToken: false` only -suppresses the projected token inside the pod; it does not detach the identity. -Cloud IAM bindings attached to that service account — IRSA on EKS, Workload -Identity on GKE — are resolved by the node/metadata path and still apply, so the -drain pod inherits the relay's cloud permissions. Create a dedicated service -account with only the object-store permissions listed below and name it -explicitly if you want the executor's IAM blast radius to be smaller than the -relay's. +relay's own service account, including cloud IAM attached through the platform's +workload-identity mechanism. For example, the EKS IRSA admission webhook can +inject its projected web-identity token and AWS environment variables despite +`automountServiceAccountToken: false`; other platforms provide their own +identity mechanism. The drain pod therefore inherits the relay's cloud role by +default. Create a dedicated service account with only the object-store +permissions listed below and name it explicitly if you want the executor's IAM +blast radius to be smaller than the relay's. The S3 principal needs the relay's normal object permissions plus bucket-level `s3:ListBucketVersions` and object-level `s3:DeleteObjectVersion` for every From 38db86befc9baac48010bc9ab48b7d184a739703 Mon Sep 17 00:00:00 2001 From: tornquist Date: Mon, 28 Sep 2026 14:46:10 +0000 Subject: [PATCH 7/9] Scope video menu probes to emitted messages Bind each context-menu interaction to the exact mock message returned by the emitter so same-second ordering cannot select the wrong video. Co-authored-by: Codex Signed-off-by: tornquist --- desktop/tests/e2e/video-attachment.spec.ts | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/desktop/tests/e2e/video-attachment.spec.ts b/desktop/tests/e2e/video-attachment.spec.ts index 2c86f5a9bda..060d356d28e 100644 --- a/desktop/tests/e2e/video-attachment.spec.ts +++ b/desktop/tests/e2e/video-attachment.spec.ts @@ -1513,12 +1513,14 @@ test("right-click menus expose distinct selectors for links, relay video, and of // ── Relay video menu: Download video + Copy link, appearing only once the // relay origin resolves (the reactivity fix) ───────────────────────────── - await emitVideoMessage(page, { + const relayMessage = (await emitVideoMessage(page, { url: MENU_RELAY_VIDEO_URL, sha: MENU_RELAY_VIDEO_SHA, filename: "relay-clip.mp4", - }); - const relayPlayer = page.getByTestId("video-player").last(); + })) as { id: string }; + const relayPlayer = page + .locator(`[data-message-id="${relayMessage.id}"]`) + .getByTestId("video-player"); await expect(relayPlayer).toBeVisible(); // Right-click the player surface. `force` skips the actionability guard: the // Play-button overlay sits above the video, but the contextmenu event still @@ -1559,12 +1561,14 @@ test("right-click menus expose distinct selectors for links, relay video, and of await expect(page.locator("[data-video-context-menu]")).toHaveCount(0); // ── Off-relay video control: renders and offers Copy link, never Download ─ - await emitVideoMessage(page, { + const offRelayMessage = (await emitVideoMessage(page, { url: MENU_OFF_RELAY_VIDEO_URL, sha: MENU_OFF_RELAY_VIDEO_SHA, filename: "external-clip.mp4", - }); - const offRelayPlayer = page.getByTestId("video-player").last(); + })) as { id: string }; + const offRelayPlayer = page + .locator(`[data-message-id="${offRelayMessage.id}"]`) + .getByTestId("video-player"); await expect(offRelayPlayer).toBeVisible(); await offRelayPlayer.click({ button: "right", force: true }); From 131ef21a253367db5e092fd6a392398e4b6eae93 Mon Sep 17 00:00:00 2001 From: Codex Date: Mon, 28 Sep 2026 20:06:50 +0000 Subject: [PATCH 8/9] test(desktop): await channel head before scroll summary check Signed-off-by: Codex Co-authored-by: Codex --- desktop/tests/e2e/scroll-history.spec.ts | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/desktop/tests/e2e/scroll-history.spec.ts b/desktop/tests/e2e/scroll-history.spec.ts index 27bf08b731f..5282c4c977e 100644 --- a/desktop/tests/e2e/scroll-history.spec.ts +++ b/desktop/tests/e2e/scroll-history.spec.ts @@ -2172,6 +2172,11 @@ test("thread summary badge survives a retained older-history prepend", async ({ await page.getByTestId("channel-deep-history").click(); await expect(page.getByTestId("chat-title")).toHaveText("deep-history"); + await waitForMockChannelHeadReady( + page, + "deep-history", + "feedf00d-0000-4000-8000-000000000007", + ); const timeline = page.getByTestId("message-timeline"); const badgeSelector = '[data-testid="message-thread-summary"][data-thread-head-id="mock-deep-history-599"]'; From 5650ae7a5dcdbbc88dedd0e4b60a3270bd9ec959 Mon Sep 17 00:00:00 2001 From: Codex Date: Mon, 28 Sep 2026 20:50:54 +0000 Subject: [PATCH 9/9] test(desktop): record transient snapshot before fallback Signed-off-by: Codex --- desktop/tests/e2e/sidebar-snapshot.spec.ts | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/desktop/tests/e2e/sidebar-snapshot.spec.ts b/desktop/tests/e2e/sidebar-snapshot.spec.ts index 6d1b349e884..bff5af61209 100644 --- a/desktop/tests/e2e/sidebar-snapshot.spec.ts +++ b/desktop/tests/e2e/sidebar-snapshot.spec.ts @@ -487,6 +487,7 @@ test("mismatched not-modified hash falls back to a full list", async ({ page, }) => { await seedSnapshot(page, { hash: "persisted-stale-hash" }); + await trackSnapshotRows(page); await installMockBridge(page, { channelsReadDelayMs: READ_DELAY_MS, channelsNotModifiedResponses: 1, @@ -494,9 +495,11 @@ test("mismatched not-modified hash falls back to a full list", async ({ await page.goto("/"); const snapshotRows = page.locator('[data-channel-id^="snapshot-"]'); - await expect(snapshotRows).toHaveCount(FULL_SNAPSHOT.length, { - timeout: 500, - }); + // Record the transient boot frame independently of test-runner scheduling; + // cold-boot coverage separately enforces the snapshot paint deadline. + await expect + .poll(() => getTrackedSnapshotRows(page)) + .toEqual(FULL_SNAPSHOT.map((channel) => channel.id)); await expect .poll(() => getChannelsPayloads(page)) .toEqual([{ knownHash: "persisted-stale-hash" }, { knownHash: null }]);