diff --git a/.github/ci/operator-postgres-values.yaml b/.github/ci/operator-postgres-values.yaml new file mode 100644 index 00000000..b4193e20 --- /dev/null +++ b/.github/ci/operator-postgres-values.yaml @@ -0,0 +1,70 @@ +# Values for the operator + Postgres E2E step in .github/workflows/test.yml, +# which installs one OpenFGA version and then upgrades to a newer one. Kept +# out of charts/openfga/ci/ because chart-testing only installs one version. +replicaCount: 1 + +datastore: + engine: postgres + uriSecret: openfga-e2e-postgres-credentials + +openfga-operator: + enabled: true + image: + pullPolicy: Never + +extraObjects: + - apiVersion: v1 + kind: Secret + metadata: + name: openfga-e2e-postgres-credentials + stringData: + uri: "postgres://openfga:changeme@openfga-e2e-postgres:5432/openfga?sslmode=disable" + - apiVersion: apps/v1 + kind: Deployment + metadata: + name: openfga-e2e-postgres + spec: + replicas: 1 + selector: + matchLabels: + app: openfga-e2e-postgres + template: + metadata: + labels: + app: openfga-e2e-postgres + spec: + containers: + - name: postgres + image: postgres:17 + ports: + - containerPort: 5432 + env: + - name: POSTGRES_USER + value: openfga + - name: POSTGRES_PASSWORD + value: changeme + - name: POSTGRES_DB + value: openfga + - name: PGDATA + value: /var/lib/postgresql/data/pgdata + volumeMounts: + - name: data + mountPath: /var/lib/postgresql/data + readinessProbe: + exec: + command: ["pg_isready", "-U", "openfga", "-d", "openfga"] + initialDelaySeconds: 5 + periodSeconds: 5 + volumes: + - name: data + emptyDir: {} + - apiVersion: v1 + kind: Service + metadata: + name: openfga-e2e-postgres + spec: + selector: + app: openfga-e2e-postgres + ports: + - port: 5432 + targetPort: 5432 diff --git a/.github/dependabot.yaml b/.github/dependabot.yaml index a561bab4..7560ce9e 100644 --- a/.github/dependabot.yaml +++ b/.github/dependabot.yaml @@ -19,3 +19,21 @@ updates: dependencies: patterns: - "*" + + - package-ecosystem: "gomod" + directory: "/operator" + schedule: + interval: "weekly" + groups: + dependencies: + patterns: + - "*" + + - package-ecosystem: "docker" + directory: "/operator" + schedule: + interval: "weekly" + groups: + dependencies: + patterns: + - "*" diff --git a/.github/scripts/check-operator-release.sh b/.github/scripts/check-operator-release.sh new file mode 100755 index 00000000..3ee58ac2 --- /dev/null +++ b/.github/scripts/check-operator-release.sh @@ -0,0 +1,61 @@ +#!/usr/bin/env bash +# Fails when operator image or chart inputs changed but the chart versions that +# publish them did not. Usage: check-operator-release.sh , e.g. origin/main. +# +# CI publishes ghcr.io/openfga/openfga-operator: once and never +# overwrites it, and chart-releaser skips chart versions that already exist. +# Release inputs are therefore only published when, in the same PR: +# 1. charts/openfga-operator/Chart.yaml bumps version +# 2. image changes also bump appVersion +# 3. charts/openfga/Chart.yaml bumps version and pins the new operator chart +# Chart.lock has to match too; `helm dependency build` in CI fails if it does not. +set -euo pipefail + +base=$1 +image_inputs=(operator/cmd operator/internal operator/go.mod operator/go.sum operator/Dockerfile) +image_changed=false +chart_changed=false +if ! git diff --quiet "$base...HEAD" -- "${image_inputs[@]}"; then + image_changed=true +fi +if ! git diff --quiet "$base...HEAD" -- charts/openfga-operator; then + chart_changed=true +fi +if [[ "$image_changed" == "false" && "$chart_changed" == "false" ]]; then + echo "operator release inputs unchanged" + exit 0 +fi + +field() { grep "^$1:" "$2" | awk '{print $2}' | tr -d '"'; } +dependency_version() { awk '/name: openfga-operator/{f=1} f && /version:/{print $2; exit}' "$1" | tr -d '"'; } +at_base() { git show "$base:$1" 2>/dev/null; } +fail() { echo "::error file=$1::$2"; exit 1; } + +operator_chart=charts/openfga-operator/Chart.yaml +parent_chart=charts/openfga/Chart.yaml +lock_file=charts/openfga/Chart.lock +operator_version=$(field version "$operator_chart") + +if at_base "$operator_chart" > /tmp/base-operator-chart.yaml; then + if [[ "$(field version /tmp/base-operator-chart.yaml)" == "$(field version "$operator_chart")" ]]; then + fail "$operator_chart" "operator release inputs changed but version is still $(field version "$operator_chart"); bump it so the chart change is published" + fi + if [[ "$image_changed" == "true" ]] && + [[ "$(field appVersion /tmp/base-operator-chart.yaml)" == "$(field appVersion "$operator_chart")" ]]; then + fail "$operator_chart" "operator image inputs changed but appVersion is still $(field appVersion "$operator_chart"); bump it so the image change is published" + fi +else + echo "operator chart is new in this PR" +fi + +at_base "$parent_chart" > /tmp/base-parent-chart.yaml +if [[ "$(field version /tmp/base-parent-chart.yaml)" == "$(field version $parent_chart)" ]]; then + fail "$parent_chart" "operator release inputs changed but the openfga chart version is still $(field version $parent_chart); bump it so a chart with the new operator is released" +fi +if [[ "$(dependency_version "$parent_chart")" != "$operator_version" ]]; then + fail "$parent_chart" "openfga chart pins openfga-operator $(dependency_version "$parent_chart") but the operator chart is $operator_version; update the dependency and run helm dependency update charts/openfga" +fi +if [[ "$(dependency_version "$lock_file")" != "$operator_version" ]]; then + fail "$lock_file" "Chart.lock pins openfga-operator $(dependency_version "$lock_file") but the operator chart is $operator_version; run helm dependency update charts/openfga" +fi +echo "operator release versions are consistent" diff --git a/.github/scripts/check-operator-release_test.sh b/.github/scripts/check-operator-release_test.sh new file mode 100755 index 00000000..d3c38ab4 --- /dev/null +++ b/.github/scripts/check-operator-release_test.sh @@ -0,0 +1,100 @@ +#!/usr/bin/env bash +set -euo pipefail + +repo=$(git rev-parse --show-toplevel) +check="$repo/.github/scripts/check-operator-release.sh" +failures=0 + +run_case() { + local name=$1 expected=$2 image_changed=$3 template_changed=$4 operator_version=$5 operator_app_version=$6 + local parent_version=$7 parent_dependency=$8 lock_dependency=$9 + local case_dir base result=0 + + case_dir=$(mktemp -d) + mkdir -p "$case_dir/operator/internal/controller" "$case_dir/charts/openfga-operator/templates" "$case_dir/charts/openfga" + ( + cd "$case_dir" + git init -q + git config user.name "Release Guard Test" + git config user.email "release-guard@example.com" + + printf 'package controller\n' > operator/internal/controller/controller.go + cat > charts/openfga-operator/Chart.yaml <<'EOF' +apiVersion: v2 +name: openfga-operator +version: "1.0.0" +appVersion: "1.0.0" +EOF + printf 'value: base\n' > charts/openfga-operator/templates/config.yaml + cat > charts/openfga/Chart.yaml <<'EOF' +apiVersion: v2 +name: openfga +version: "1.0.0" +dependencies: + - name: openfga-operator + version: "1.0.0" +EOF + cat > charts/openfga/Chart.lock <<'EOF' +dependencies: +- name: openfga-operator + version: 1.0.0 +EOF + git add . + git commit -qm base + base=$(git rev-parse HEAD) + + if [[ "$image_changed" == "true" ]]; then + printf 'var changed = true\n' >> operator/internal/controller/controller.go + fi + if [[ "$template_changed" == "true" ]]; then + printf 'value: candidate\n' > charts/openfga-operator/templates/config.yaml + fi + cat > charts/openfga-operator/Chart.yaml < charts/openfga/Chart.yaml < charts/openfga/Chart.lock </dev/null 2>&1 || result=$? + if [[ "$expected" == "pass" && "$result" -ne 0 ]] || + [[ "$expected" == "fail" && "$result" -eq 0 ]]; then + printf 'FAIL: %s expected %s, exit code %d\n' "$name" "$expected" "$result" + exit 1 + fi + ) || failures=$((failures + 1)) +} + +run_case "unchanged release inputs" pass false false 1.0.0 1.0.0 1.0.0 1.0.0 1.0.0 +run_case "consistent image release" pass true false 1.1.0 1.1.0 1.1.0 1.1.0 1.1.0 +run_case "operator version unchanged" fail true false 1.0.0 1.1.0 1.1.0 1.0.0 1.0.0 +run_case "operator appVersion unchanged" fail true false 1.1.0 1.0.0 1.1.0 1.1.0 1.1.0 +run_case "parent version unchanged" fail true false 1.1.0 1.1.0 1.0.0 1.1.0 1.1.0 +run_case "parent dependency mismatch" fail true false 1.1.0 1.1.0 1.1.0 1.0.0 1.1.0 +run_case "lock dependency mismatch" fail true false 1.1.0 1.1.0 1.1.0 1.1.0 1.0.0 +run_case "consistent chart-only release" pass false true 1.1.0 1.0.0 1.1.0 1.1.0 1.1.0 +run_case "chart-only version unchanged" fail false true 1.0.0 1.0.0 1.1.0 1.0.0 1.0.0 + +if [[ "$failures" -ne 0 ]]; then + printf '%d release guard case(s) failed\n' "$failures" + exit 1 +fi + +echo "operator release guard matrix passed" diff --git a/.github/workflows/operator.yml b/.github/workflows/operator.yml new file mode 100644 index 00000000..9a2ae680 --- /dev/null +++ b/.github/workflows/operator.yml @@ -0,0 +1,183 @@ +name: Operator + +on: + push: + branches: + - main + paths: + - "operator/**" + - "charts/openfga-operator/**" + - ".github/scripts/check-operator-release*.sh" + - ".github/workflows/operator.yml" + pull_request: + paths: + - "operator/**" + - "charts/openfga-operator/**" + - ".github/scripts/check-operator-release*.sh" + - ".github/workflows/operator.yml" + workflow_dispatch: + inputs: + push_image: + description: "Push the operator image to GHCR" + type: boolean + default: true + +env: + IMAGE_NAME: ghcr.io/${{ github.repository_owner }}/openfga-operator + +jobs: + test: + runs-on: ubuntu-latest + permissions: + contents: read + steps: + - name: Checkout + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + fetch-depth: 0 + + - name: Require a version bump when the operator image changes + if: github.event_name == 'pull_request' + run: .github/scripts/check-operator-release.sh "origin/${{ github.base_ref }}" + + - name: Test operator release guard + run: .github/scripts/check-operator-release_test.sh + + - name: Set up Go + uses: actions/setup-go@40f1582b2485089dde7abd97c1529aa768e1baff # v5.6.0 + with: + go-version-file: operator/go.mod + cache-dependency-path: operator/go.sum + + - name: Run tests + working-directory: operator + run: go test ./... -race -v + + - name: Check formatting + working-directory: operator + run: test -z "$(gofmt -l .)" + + - name: Run vet + working-directory: operator + run: go vet ./... + + build-and-push: + needs: test + runs-on: ubuntu-latest + permissions: + contents: read + packages: write + id-token: write + steps: + - name: Checkout + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + + - name: Extract version from Chart.yaml + id: version + run: | + version=$(grep '^appVersion:' charts/openfga-operator/Chart.yaml | awk '{print $2}' | tr -d '"') + echo "version=${version}" >> "$GITHUB_OUTPUT" + short_sha="${GITHUB_SHA::7}" + echo "short_sha=${short_sha}" >> "$GITHUB_OUTPUT" + echo "Operator version: ${version} (sha: ${short_sha})" + + - name: Determine push policy + id: policy + run: | + if [[ "$GITHUB_EVENT_NAME" == "push" && "$GITHUB_REF" == "refs/heads/main" ]]; then + echo "push=true" >> "$GITHUB_OUTPUT" + echo "mode=main" >> "$GITHUB_OUTPUT" + elif [[ "$GITHUB_EVENT_NAME" == "workflow_dispatch" && "${{ inputs.push_image }}" == "true" ]]; then + echo "push=true" >> "$GITHUB_OUTPUT" + echo "mode=dispatch" >> "$GITHUB_OUTPUT" + else + echo "push=false" >> "$GITHUB_OUTPUT" + echo "mode=pr" >> "$GITHUB_OUTPUT" + fi + + - name: Set up QEMU + uses: docker/setup-qemu-action@c7c53464625b32c7a7e944ae62b3e17d2b600130 # v3.7.0 + + - name: Set up Docker Buildx + uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0 + + - name: Login to GHCR + if: steps.policy.outputs.push == 'true' + uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0 + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + + - name: Determine image tags + id: tags + run: | + set -euo pipefail + version="${{ steps.version.outputs.version }}" + sha="${{ steps.version.outputs.short_sha }}" + img="${{ env.IMAGE_NAME }}" + case "${{ steps.policy.outputs.mode }}" in + main) + # Always refresh :latest and publish the immutable per-commit tag. + tags="${img}:latest,${img}:${version}-${sha}" + # Publish : only if it is not already in the registry, so a + # released version is never overwritten by a later push to main. This + # mirrors chart-releaser's CR_SKIP_EXISTING for the charts and keeps + # the chart's default image tag stable for a given version. + if docker buildx imagetools inspect "${img}:${version}" >/dev/null 2>&1; then + echo "::notice::${img}:${version} already exists; leaving it unchanged" + else + tags="${img}:${version},${tags}" + fi + ;; + dispatch) + # Manual run: publish only the immutable per-commit tag. + tags="${img}:${version}-${sha}" + ;; + *) + # Pull request: build both platforms but do not publish — catches + # arm64-incompatible changes (build tags, syscalls, CGO) before merge. + tags="${img}:pr-${sha}" + ;; + esac + echo "tags=${tags}" >> "$GITHUB_OUTPUT" + echo "Resolved tags: ${tags}" + + - name: Build and (conditionally) push + id: build + uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2 + with: + context: operator + push: ${{ steps.policy.outputs.push }} + platforms: linux/amd64,linux/arm64 + tags: ${{ steps.tags.outputs.tags }} + # SBOM and build provenance are attached to the pushed image index. + sbom: ${{ steps.policy.outputs.push == 'true' }} + provenance: ${{ steps.policy.outputs.push == 'true' && 'mode=max' || 'false' }} + cache-from: type=gha + cache-to: type=gha,mode=max + labels: | + org.opencontainers.image.source=https://github.com/${{ github.repository }} + org.opencontainers.image.version=${{ steps.version.outputs.version }} + org.opencontainers.image.revision=${{ github.sha }} + org.opencontainers.image.title=openfga-operator + org.opencontainers.image.description=OpenFGA Kubernetes operator for migration orchestration + + - name: Install cosign + if: steps.policy.outputs.push == 'true' + uses: sigstore/cosign-installer@6f9f17788090df1f26f669e9d70d6ae9567deba6 # v4.1.2 + with: + cosign-release: "v2.6.1" + + # Keyless signature on the digest covers every tag pushed above. + - name: Sign and verify image + if: steps.policy.outputs.push == 'true' + env: + IMAGE: ${{ env.IMAGE_NAME }}@${{ steps.build.outputs.digest }} + IDENTITY: https://github.com/${{ github.repository }}/.github/workflows/operator.yml@${{ github.ref }} + run: | + cosign sign --yes "$IMAGE" + cosign verify "$IMAGE" \ + --certificate-oidc-issuer https://token.actions.githubusercontent.com \ + --certificate-identity "$IDENTITY" >/dev/null + echo "signed and verified $IMAGE" diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index ff667113..04d82919 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -24,6 +24,7 @@ jobs: helm repo add bitnami-legacy https://raw.githubusercontent.com/bitnami/charts/archive-full-index/bitnami helm dependency build charts/openfga helm unittest charts/openfga + helm unittest charts/openfga-operator - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: @@ -37,7 +38,10 @@ jobs: id: list-changed run: | changed=$(ct list-changed --target-branch ${{ github.event.repository.default_branch }}) - if [[ -n "$changed" ]]; then + # Operator code changes are only exercised against a cluster by the + # steps below, so treat them as a chart change too. + operator=$(git diff --name-only "origin/${{ github.event.repository.default_branch }}...HEAD" -- operator) + if [[ -n "$changed" || -n "$operator" ]]; then echo "changed=true" >> "$GITHUB_OUTPUT" fi @@ -59,6 +63,96 @@ jobs: if: steps.list-changed.outputs.changed == 'true' uses: helm/kind-action@06c1ae10762d3b9c1644e7fe69596ae519e015a2 # v1.15.0 + - name: Build and load operator image into kind + if: steps.list-changed.outputs.changed == 'true' + run: | + version=$(grep '^appVersion:' charts/openfga-operator/Chart.yaml | awk '{print $2}' | tr -d '"') + docker build -t "ghcr.io/openfga/openfga-operator:${version}" operator/ + kind load docker-image "ghcr.io/openfga/openfga-operator:${version}" --name chart-testing + - name: Run chart-testing (install) if: steps.list-changed.outputs.changed == 'true' run: ct install --target-branch ${{ github.event.repository.default_branch }} + + - name: E2E test — operator-managed migration across schema boundary + id: e2e-operator + if: steps.list-changed.outputs.changed == 'true' + env: + NS: openfga-e2e + REL: openfga + # v1.9.5 predates the v1.10.0 "!!REQUIRES MIGRATION!!" boundary + # (collation spec change in openfga/openfga#2661), so upgrading to the + # chart's appVersion always crosses at least one migration. + OLD_VER: v1.9.5 + run: | + set -euo pipefail + NEW_VER=$(grep '^appVersion:' charts/openfga/Chart.yaml | awk '{print $2}' | tr -d '"') + kubectl create namespace "$NS" + helm dependency build charts/openfga + + echo "=== Phase 1: fresh install at ${OLD_VER} ===" + helm install "$REL" charts/openfga \ + --namespace "$NS" \ + --values .github/ci/operator-postgres-values.yaml \ + --set image.tag="${OLD_VER}" \ + --wait --timeout=3m + + # Operator pod must reach Ready (validates /readyz, RBAC, env vars). + kubectl wait deployment -n "$NS" \ + -l app.kubernetes.io/name=openfga-operator \ + --for=condition=Available=True --timeout=2m + + # Operator must run the migration Job and write ConfigMap at OLD_VER. + # Poll because kubectl wait --for=create requires kubectl >=1.31. + for _ in $(seq 1 60); do + ver=$(kubectl get configmap "${REL}-migration-status" -n "$NS" \ + -o jsonpath='{.data.version}' 2>/dev/null || true) + if [ "$ver" = "${OLD_VER}" ]; then + echo "Phase 1: migration ConfigMap version=${ver}" + break + fi + sleep 3 + done + test "$ver" = "${OLD_VER}" + + # The pod stays NotReady until the migration has run on the new database. + kubectl wait deployment/"$REL" -n "$NS" \ + --for=jsonpath='{.status.readyReplicas}'=1 --timeout=3m + + echo "=== Phase 2: helm upgrade ${OLD_VER} → ${NEW_VER} ===" + helm upgrade "$REL" charts/openfga \ + --namespace "$NS" \ + --values .github/ci/operator-postgres-values.yaml \ + --set image.tag="${NEW_VER}" \ + --wait --timeout=3m + + # Operator must detect the version change, delete the stale Job, + # run a new migration, and update the ConfigMap to NEW_VER. + for _ in $(seq 1 60); do + ver=$(kubectl get configmap "${REL}-migration-status" -n "$NS" \ + -o jsonpath='{.data.version}' 2>/dev/null || true) + if [ "$ver" = "${NEW_VER}" ]; then + echo "Phase 2: migration ConfigMap version=${ver}" + break + fi + sleep 3 + done + test "$ver" = "${NEW_VER}" + + # New pods must roll out at NEW_VER and become Ready. + kubectl wait deployment/"$REL" -n "$NS" \ + --for=jsonpath='{.status.readyReplicas}'=1 --timeout=3m + image=$(kubectl get deployment/"$REL" -n "$NS" \ + -o jsonpath='{.spec.template.spec.containers[0].image}') + echo "Phase 2 running image: $image" + echo "$image" | grep -q ":${NEW_VER}" + + - name: Dump operator E2E diagnostics on failure + if: failure() && steps.e2e-operator.conclusion == 'failure' + env: + NS: openfga-e2e + run: | + kubectl get all,configmap,job -n "$NS" -o wide || true + kubectl describe deployment -n "$NS" || true + kubectl logs -n "$NS" -l app.kubernetes.io/name=openfga-operator --tail=200 || true + kubectl logs -n "$NS" -l job-name --tail=200 || true diff --git a/README.md b/README.md index ad9446ee..5721bce4 100644 --- a/README.md +++ b/README.md @@ -14,6 +14,7 @@ It is designed to make it easy for developers to model their application permiss ## Charts * [openfga](https://github.com/openfga/helm-charts/blob/main/charts/openfga) +* [openfga-operator](https://github.com/openfga/helm-charts/blob/main/charts/openfga-operator) — runs OpenFGA database migrations; installed by the openfga chart with `openfga-operator.enabled: true` ## Contributing diff --git a/charts/openfga-operator/.helmignore b/charts/openfga-operator/.helmignore new file mode 100644 index 00000000..edf9e7ef --- /dev/null +++ b/charts/openfga-operator/.helmignore @@ -0,0 +1,18 @@ +# Patterns to ignore when building packages. +.DS_Store +.git +.gitignore +.bzr +.bzrignore +.hg +.hgignore +.svn +*.swp +*.bak +*.tmp +*.orig +*~ +.project +.idea +*.tmproj +.vscode diff --git a/charts/openfga-operator/Chart.yaml b/charts/openfga-operator/Chart.yaml new file mode 100644 index 00000000..0a321ac8 --- /dev/null +++ b/charts/openfga-operator/Chart.yaml @@ -0,0 +1,24 @@ +apiVersion: v2 +name: openfga-operator +description: Helm chart for the OpenFGA Kubernetes operator. + +type: application +version: 0.1.0 +appVersion: "0.1.0" + +home: "https://openfga.github.io/helm-charts" +icon: https://github.com/openfga/community/raw/main/brand-assets/icon/color/openfga-icon-color.svg + +maintainers: + - name: OpenFGA Authors + url: https://github.com/openfga +sources: + - https://github.com/openfga/helm-charts + +annotations: + artifacthub.io/license: Apache-2.0 + artifacthub.io/operator: "true" + artifacthub.io/operatorCapabilities: Basic Install + artifacthub.io/signKey: | + fingerprint: 8E9B315F6C22E339959DA77B35CCF4BDC9F58F2A + url: https://openfga.github.io/helm-charts/pgp-public-key.asc diff --git a/charts/openfga-operator/ci/default-values.yaml b/charts/openfga-operator/ci/default-values.yaml new file mode 100644 index 00000000..93797cd5 --- /dev/null +++ b/charts/openfga-operator/ci/default-values.yaml @@ -0,0 +1,4 @@ +# Standalone install exercise for chart-testing. +# kind has the operator image preloaded, so skip the registry pull. +image: + pullPolicy: Never diff --git a/charts/openfga-operator/templates/NOTES.txt b/charts/openfga-operator/templates/NOTES.txt new file mode 100644 index 00000000..cead36dc --- /dev/null +++ b/charts/openfga-operator/templates/NOTES.txt @@ -0,0 +1,13 @@ +The openfga-operator has been deployed. + +To check operator status: + kubectl get deployment --namespace {{ include "openfga-operator.namespace" . }} {{ include "openfga-operator.fullname" . }} + +To view operator logs: + kubectl logs --namespace {{ include "openfga-operator.namespace" . }} -l "app.kubernetes.io/name={{ include "openfga-operator.name" . }}" + +To check migration status: + kubectl get configmap -n {{ include "openfga-operator.watchNamespace" . }} -l app.kubernetes.io/managed-by=openfga-operator + +To inspect migration jobs: + kubectl get jobs -n {{ include "openfga-operator.watchNamespace" . }} -l app.kubernetes.io/part-of=openfga,app.kubernetes.io/component=migration diff --git a/charts/openfga-operator/templates/_helpers.tpl b/charts/openfga-operator/templates/_helpers.tpl new file mode 100644 index 00000000..dc7f512c --- /dev/null +++ b/charts/openfga-operator/templates/_helpers.tpl @@ -0,0 +1,79 @@ +{{/* +Expand the name of the chart. +*/}} +{{- define "openfga-operator.name" -}} +{{- default .Chart.Name .Values.nameOverride | trunc 63 | trimSuffix "-" }} +{{- end }} + +{{/* +Create a default fully qualified app name. +We truncate at 63 chars because some Kubernetes name fields are limited to this (by the DNS naming spec). +If release name contains chart name it will be used as a full name. +*/}} +{{- define "openfga-operator.fullname" -}} +{{- if .Values.fullnameOverride }} +{{- .Values.fullnameOverride | trunc 63 | trimSuffix "-" }} +{{- else }} +{{- $name := default .Chart.Name .Values.nameOverride }} +{{- if contains $name .Release.Name }} +{{- .Release.Name | trunc 63 | trimSuffix "-" }} +{{- else }} +{{- printf "%s-%s" .Release.Name $name | trunc 63 | trimSuffix "-" }} +{{- end }} +{{- end }} +{{- end }} + +{{/* +Expand the namespace of the release. +Allows overriding it for multi-namespace deployments in combined charts. +*/}} +{{- define "openfga-operator.namespace" -}} +{{- default .Release.Namespace .Values.namespaceOverride | trunc 63 | trimSuffix "-" -}} +{{- end -}} + +{{/* +Expand the namespace watched and managed by the operator. +*/}} +{{- define "openfga-operator.watchNamespace" -}} +{{- default (include "openfga-operator.namespace" .) .Values.watchNamespace | trunc 63 | trimSuffix "-" -}} +{{- end -}} + +{{/* +Create chart name and version as used by the chart label. +*/}} +{{- define "openfga-operator.chart" -}} +{{- printf "%s-%s" .Chart.Name .Chart.Version | replace "+" "_" | trunc 63 | trimSuffix "-" }} +{{- end }} + +{{/* +Common labels +*/}} +{{- define "openfga-operator.labels" -}} +helm.sh/chart: {{ include "openfga-operator.chart" . }} +{{ include "openfga-operator.selectorLabels" . }} +app.kubernetes.io/component: operator +{{- if .Chart.AppVersion }} +app.kubernetes.io/version: {{ .Chart.AppVersion | quote }} +{{- end }} +app.kubernetes.io/managed-by: {{ .Release.Service }} +app.kubernetes.io/part-of: openfga +{{- end }} + +{{/* +Selector labels +*/}} +{{- define "openfga-operator.selectorLabels" -}} +app.kubernetes.io/name: {{ include "openfga-operator.name" . }} +app.kubernetes.io/instance: {{ .Release.Name }} +{{- end }} + +{{/* +Create the name of the service account to use +*/}} +{{- define "openfga-operator.serviceAccountName" -}} +{{- if .Values.serviceAccount.create }} +{{- default (include "openfga-operator.fullname" .) .Values.serviceAccount.name }} +{{- else }} +{{- required "serviceAccount.name must be set when serviceAccount.create=false" .Values.serviceAccount.name }} +{{- end }} +{{- end }} diff --git a/charts/openfga-operator/templates/deployment.yaml b/charts/openfga-operator/templates/deployment.yaml new file mode 100644 index 00000000..e2f22189 --- /dev/null +++ b/charts/openfga-operator/templates/deployment.yaml @@ -0,0 +1,91 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: {{ include "openfga-operator.fullname" . }} + namespace: {{ include "openfga-operator.namespace" . }} + labels: + {{- include "openfga-operator.labels" . | nindent 4 }} +spec: + replicas: {{ .Values.replicaCount }} + selector: + matchLabels: + {{- include "openfga-operator.selectorLabels" . | nindent 6 }} + template: + metadata: + {{- with .Values.podAnnotations }} + annotations: + {{- toYaml . | nindent 8 }} + {{- end }} + labels: + {{- include "openfga-operator.selectorLabels" . | nindent 8 }} + spec: + {{- with .Values.imagePullSecrets }} + imagePullSecrets: + {{- toYaml . | nindent 8 }} + {{- end }} + serviceAccountName: {{ include "openfga-operator.serviceAccountName" . }} + {{- with .Values.podSecurityContext }} + securityContext: + {{- toYaml . | nindent 8 }} + {{- end }} + containers: + - name: operator + {{- with .Values.securityContext }} + securityContext: + {{- toYaml . | nindent 12 }} + {{- end }} + image: "{{ .Values.image.repository }}:{{ .Values.image.tag | default .Chart.AppVersion }}" + imagePullPolicy: {{ .Values.image.pullPolicy }} + args: + {{- if .Values.leaderElection.enabled }} + - --leader-elect + {{- end }} + {{- if .Values.watchNamespace }} + - --watch-namespace={{ .Values.watchNamespace }} + {{- end }} + - --metrics-bind-address={{ if .Values.metrics.enabled }}:8080{{ else }}0{{ end }} + - --backoff-limit={{ .Values.migrationJob.backoffLimit }} + - --active-deadline-seconds={{ .Values.migrationJob.activeDeadlineSeconds }} + - --ttl-seconds-after-finished={{ .Values.migrationJob.ttlSecondsAfterFinished }} + env: + - name: POD_NAMESPACE + valueFrom: + fieldRef: + fieldPath: metadata.namespace + ports: + - name: healthz + containerPort: 8081 + protocol: TCP + {{- if .Values.metrics.enabled }} + - name: metrics + containerPort: 8080 + protocol: TCP + {{- end }} + livenessProbe: + httpGet: + path: /healthz + port: healthz + initialDelaySeconds: 15 + periodSeconds: 20 + readinessProbe: + httpGet: + path: /readyz + port: healthz + initialDelaySeconds: 5 + periodSeconds: 10 + {{- with .Values.resources }} + resources: + {{- toYaml . | nindent 12 }} + {{- end }} + {{- with .Values.nodeSelector }} + nodeSelector: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.affinity }} + affinity: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- with .Values.tolerations }} + tolerations: + {{- toYaml . | nindent 8 }} + {{- end }} diff --git a/charts/openfga-operator/templates/pdb.yaml b/charts/openfga-operator/templates/pdb.yaml new file mode 100644 index 00000000..9a068a3c --- /dev/null +++ b/charts/openfga-operator/templates/pdb.yaml @@ -0,0 +1,18 @@ +{{- if .Values.podDisruptionBudget.enabled -}} +apiVersion: policy/v1 +kind: PodDisruptionBudget +metadata: + name: {{ include "openfga-operator.fullname" . }} + namespace: {{ include "openfga-operator.namespace" . }} + labels: + {{- include "openfga-operator.labels" . | nindent 4 }} +spec: + {{- if ne (toString .Values.podDisruptionBudget.minAvailable) "" }} + minAvailable: {{ .Values.podDisruptionBudget.minAvailable }} + {{- else }} + maxUnavailable: {{ .Values.podDisruptionBudget.maxUnavailable }} + {{- end }} + selector: + matchLabels: + {{- include "openfga-operator.selectorLabels" . | nindent 6 }} +{{- end }} diff --git a/charts/openfga-operator/templates/role.yaml b/charts/openfga-operator/templates/role.yaml new file mode 100644 index 00000000..eb771c30 --- /dev/null +++ b/charts/openfga-operator/templates/role.yaml @@ -0,0 +1,28 @@ +{{- if .Values.rbac.create -}} +apiVersion: rbac.authorization.k8s.io/v1 +kind: Role +metadata: + name: {{ include "openfga-operator.fullname" . }} + namespace: {{ include "openfga-operator.watchNamespace" . }} + labels: + {{- include "openfga-operator.labels" . | nindent 4 }} +rules: + - apiGroups: ["apps"] + resources: ["deployments"] + verbs: ["get", "list", "watch"] + - apiGroups: ["apps"] + resources: ["deployments/status"] + verbs: ["patch"] + - apiGroups: ["batch"] + resources: ["jobs"] + verbs: ["get", "list", "watch", "create", "delete", "patch"] + - apiGroups: [""] + resources: ["configmaps"] + verbs: ["get", "list", "watch", "create", "update"] + - apiGroups: ["coordination.k8s.io"] + resources: ["leases"] + verbs: ["get", "list", "watch", "create", "update"] + - apiGroups: [""] + resources: ["events"] + verbs: ["create", "patch"] +{{- end }} diff --git a/charts/openfga-operator/templates/rolebinding.yaml b/charts/openfga-operator/templates/rolebinding.yaml new file mode 100644 index 00000000..d892c2c6 --- /dev/null +++ b/charts/openfga-operator/templates/rolebinding.yaml @@ -0,0 +1,17 @@ +{{- if .Values.rbac.create -}} +apiVersion: rbac.authorization.k8s.io/v1 +kind: RoleBinding +metadata: + name: {{ include "openfga-operator.fullname" . }} + namespace: {{ include "openfga-operator.watchNamespace" . }} + labels: + {{- include "openfga-operator.labels" . | nindent 4 }} +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: Role + name: {{ include "openfga-operator.fullname" . }} +subjects: + - kind: ServiceAccount + name: {{ include "openfga-operator.serviceAccountName" . }} + namespace: {{ include "openfga-operator.namespace" . }} +{{- end }} diff --git a/charts/openfga-operator/templates/serviceaccount.yaml b/charts/openfga-operator/templates/serviceaccount.yaml new file mode 100644 index 00000000..8b1f8941 --- /dev/null +++ b/charts/openfga-operator/templates/serviceaccount.yaml @@ -0,0 +1,13 @@ +{{- if .Values.serviceAccount.create -}} +apiVersion: v1 +kind: ServiceAccount +metadata: + name: {{ include "openfga-operator.serviceAccountName" . }} + namespace: {{ include "openfga-operator.namespace" . }} + labels: + {{- include "openfga-operator.labels" . | nindent 4 }} + {{- with .Values.serviceAccount.annotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} +{{- end }} diff --git a/charts/openfga-operator/tests/metrics_test.yaml b/charts/openfga-operator/tests/metrics_test.yaml new file mode 100644 index 00000000..3d673809 --- /dev/null +++ b/charts/openfga-operator/tests/metrics_test.yaml @@ -0,0 +1,29 @@ +suite: operator metrics +templates: + - templates/deployment.yaml +tests: + - it: should disable the metrics endpoint by default + asserts: + - contains: + path: spec.template.spec.containers[0].args + content: --metrics-bind-address=0 + - notContains: + path: spec.template.spec.containers[0].ports + content: + name: metrics + containerPort: 8080 + protocol: TCP + + - it: should serve metrics on a named port when enabled + set: + metrics.enabled: true + asserts: + - contains: + path: spec.template.spec.containers[0].args + content: --metrics-bind-address=:8080 + - contains: + path: spec.template.spec.containers[0].ports + content: + name: metrics + containerPort: 8080 + protocol: TCP diff --git a/charts/openfga-operator/tests/pdb_test.yaml b/charts/openfga-operator/tests/pdb_test.yaml new file mode 100644 index 00000000..465beb34 --- /dev/null +++ b/charts/openfga-operator/tests/pdb_test.yaml @@ -0,0 +1,25 @@ +suite: operator PodDisruptionBudget +templates: + - templates/pdb.yaml +tests: + - it: should preserve zero minAvailable + set: + podDisruptionBudget.enabled: true + podDisruptionBudget.minAvailable: 0 + asserts: + - equal: + path: spec.minAvailable + value: 0 + - isNull: + path: spec.maxUnavailable + + - it: should preserve zero maxUnavailable + set: + podDisruptionBudget.enabled: true + podDisruptionBudget.maxUnavailable: 0 + asserts: + - equal: + path: spec.maxUnavailable + value: 0 + - isNull: + path: spec.minAvailable diff --git a/charts/openfga-operator/tests/rbac_test.yaml b/charts/openfga-operator/tests/rbac_test.yaml new file mode 100644 index 00000000..c61de222 --- /dev/null +++ b/charts/openfga-operator/tests/rbac_test.yaml @@ -0,0 +1,16 @@ +suite: operator RBAC +templates: + - templates/role.yaml + - templates/rolebinding.yaml +tests: + - it: should create the Role and RoleBinding by default + asserts: + - hasDocuments: + count: 1 + + - it: should create nothing when rbac.create is false + set: + rbac.create: false + asserts: + - hasDocuments: + count: 0 diff --git a/charts/openfga-operator/tests/watch_namespace_role_test.yaml b/charts/openfga-operator/tests/watch_namespace_role_test.yaml new file mode 100644 index 00000000..4cc98481 --- /dev/null +++ b/charts/openfga-operator/tests/watch_namespace_role_test.yaml @@ -0,0 +1,27 @@ +suite: operator watch namespace Role +templates: + - templates/role.yaml +tests: + - it: should create managed-resource RBAC in the watch namespace + release: + namespace: operator-system + set: + watchNamespace: openfga-app + asserts: + - equal: + path: metadata.namespace + value: openfga-app + - contains: + path: rules + content: + apiGroups: + - batch + resources: + - jobs + verbs: + - get + - list + - watch + - create + - delete + - patch diff --git a/charts/openfga-operator/tests/watch_namespace_rolebinding_test.yaml b/charts/openfga-operator/tests/watch_namespace_rolebinding_test.yaml new file mode 100644 index 00000000..258830e4 --- /dev/null +++ b/charts/openfga-operator/tests/watch_namespace_rolebinding_test.yaml @@ -0,0 +1,16 @@ +suite: operator watch namespace RoleBinding +templates: + - templates/rolebinding.yaml +tests: + - it: should bind the operator service account in the watch namespace + release: + namespace: operator-system + set: + watchNamespace: openfga-app + asserts: + - equal: + path: metadata.namespace + value: openfga-app + - equal: + path: subjects[0].namespace + value: operator-system diff --git a/charts/openfga-operator/values.schema.json b/charts/openfga-operator/values.schema.json new file mode 100644 index 00000000..70f877d4 --- /dev/null +++ b/charts/openfga-operator/values.schema.json @@ -0,0 +1,125 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "properties": { + "global": { + "type": "object" + }, + "enabled": { "type": "boolean" }, + "replicaCount": { + "type": "integer", + "minimum": 1 + }, + "image": { + "type": "object", + "properties": { + "repository": { + "type": "string", + "minLength": 1 + }, + "pullPolicy": { + "type": "string", + "enum": ["Always", "IfNotPresent", "Never"] + }, + "tag": { + "type": "string" + } + }, + "required": ["repository"], + "additionalProperties": false + }, + "imagePullSecrets": { + "type": "array", + "items": { + "type": "object", + "properties": { + "name": { "type": "string" } + }, + "required": ["name"], + "additionalProperties": false + } + }, + "nameOverride": { "type": "string" }, + "fullnameOverride": { "type": "string" }, + "namespaceOverride": { "type": "string" }, + "serviceAccount": { + "type": "object", + "properties": { + "create": { "type": "boolean" }, + "annotations": { "type": "object" }, + "name": { "type": "string" } + }, + "additionalProperties": false + }, + "rbac": { + "type": "object", + "properties": { + "create": { "type": "boolean" } + }, + "additionalProperties": false + }, + "podAnnotations": { "type": "object" }, + "podSecurityContext": { "type": "object" }, + "securityContext": { "type": "object" }, + "watchNamespace": { "type": "string" }, + "leaderElection": { + "type": "object", + "properties": { + "enabled": { "type": "boolean" } + }, + "additionalProperties": false + }, + "metrics": { + "type": "object", + "properties": { + "enabled": { "type": "boolean" } + }, + "additionalProperties": false + }, + "migrationJob": { + "type": "object", + "properties": { + "backoffLimit": { + "type": "integer", + "minimum": 0 + }, + "activeDeadlineSeconds": { + "type": "integer", + "minimum": 0 + }, + "ttlSecondsAfterFinished": { + "type": "integer", + "minimum": 0 + } + }, + "additionalProperties": false + }, + "resources": { "type": "object" }, + "podDisruptionBudget": { + "type": "object", + "properties": { + "enabled": { "type": "boolean" }, + "minAvailable": { + "oneOf": [ + { "type": "string" }, + { "type": "integer", "minimum": 0 } + ] + }, + "maxUnavailable": { + "oneOf": [ + { "type": "string" }, + { "type": "integer", "minimum": 0 } + ] + } + }, + "additionalProperties": false + }, + "nodeSelector": { "type": "object" }, + "tolerations": { + "type": "array", + "items": { "type": "object" } + }, + "affinity": { "type": "object" } + }, + "additionalProperties": false +} diff --git a/charts/openfga-operator/values.yaml b/charts/openfga-operator/values.yaml new file mode 100644 index 00000000..614c9298 --- /dev/null +++ b/charts/openfga-operator/values.yaml @@ -0,0 +1,103 @@ +# -- Used as the dependency condition by the openfga chart; has no effect when +# this chart is installed on its own. +enabled: true + +replicaCount: 1 + +image: + repository: ghcr.io/openfga/openfga-operator + pullPolicy: IfNotPresent + # -- Overrides the image tag (defaults to the chart appVersion). CI publishes the + # appVersion tag once per version and does not overwrite it, so it stays stable + # for a given chart version. To pin a specific main build instead, use the + # immutable `-` tag or a digest. + tag: "" + +imagePullSecrets: [] +nameOverride: "" +fullnameOverride: "" +# -- Override the namespace for all operator resources. +# Useful when the parent chart deploys subcharts into a different namespace. +namespaceOverride: "" + +serviceAccount: + # -- Specifies whether a service account should be created. + create: true + # -- Annotations to add to the service account. + annotations: {} + # -- The name of the service account to use. + # If not set and create is true, a name is generated using the fullname template. + name: "" + +rbac: + # -- Create the Role and RoleBinding the operator needs in the watched namespace. + create: true + +podAnnotations: {} + +podSecurityContext: + runAsNonRoot: true + seccompProfile: + type: RuntimeDefault + +securityContext: + allowPrivilegeEscalation: false + capabilities: + drop: + - ALL + readOnlyRootFilesystem: true + runAsNonRoot: true + runAsUser: 65532 + +# -- Namespace to watch for OpenFGA Deployments. +# Leave empty to default to the operator pod's own namespace (read from +# the POD_NAMESPACE env var, set via the downward API). This usually +# equals the release namespace, but when `namespaceOverride` puts the +# operator in a different namespace than the release, the watch follows +# the pod — not the release. Set this explicitly to watch a specific +# namespace independent of where the operator runs. The target namespace +# must exist before installation so Helm can create the Role and RoleBinding. +watchNamespace: "" + +leaderElection: + # -- Enable leader election for controller manager. + enabled: true + +metrics: + # -- Serve controller-runtime metrics on a `metrics` port (8080). Plain HTTP + # without authentication, so leave off unless something scrapes it. + enabled: false + +migrationJob: + # -- Number of pod failures before a migration Job is considered failed. + backoffLimit: 3 + # -- Maximum wall-clock seconds a migration Job can run before being terminated. + # 0 disables the deadline. A migration that is cut off, such as an index build + # on a large table, has to start over on the next attempt. + activeDeadlineSeconds: 0 + # -- Seconds to keep completed Job pods for log inspection before garbage collection. + # Failed Jobs are retained for the operator's fixed 60-second retry delay. + ttlSecondsAfterFinished: 300 + +resources: + requests: + cpu: 10m + memory: 64Mi + limits: + memory: 128Mi + +podDisruptionBudget: + # -- Enable a PodDisruptionBudget for the operator. + enabled: false + # -- Minimum number of pods that must be available during disruption. + # Cannot be set together with maxUnavailable. + minAvailable: "" + # -- Maximum number of pods that can be unavailable during disruption. + # Defaults to 1 when enabled and minAvailable is not set. + maxUnavailable: 1 + +nodeSelector: {} + +tolerations: [] + +affinity: {} diff --git a/charts/openfga/Chart.lock b/charts/openfga/Chart.lock index e82ffa5a..ac8d7f30 100644 --- a/charts/openfga/Chart.lock +++ b/charts/openfga/Chart.lock @@ -8,5 +8,8 @@ dependencies: - name: common repository: oci://registry-1.docker.io/bitnamicharts version: 2.13.3 -digest: sha256:4bbfb25821b0dfb6c70aabb5caf4c5ec7e6526261f93a8f531f507f1d4c43e3e -generated: "2026-03-18T11:41:40.1785546-04:00" +- name: openfga-operator + repository: file://../openfga-operator + version: 0.1.0 +digest: sha256:3df1161dfa820406918b40bb5bada3da6d156a65631f29d72840087d9063db2e +generated: "2026-09-23T14:52:08.781111+05:30" diff --git a/charts/openfga/Chart.yaml b/charts/openfga/Chart.yaml index d3ebfb05..f299ff30 100644 --- a/charts/openfga/Chart.yaml +++ b/charts/openfga/Chart.yaml @@ -3,7 +3,7 @@ name: openfga description: A Kubernetes Helm chart for the OpenFGA project. type: application -version: 0.3.15 +version: 0.4.0 appVersion: "v1.21.0" home: "https://openfga.github.io/helm-charts" @@ -29,3 +29,7 @@ dependencies: repository: oci://registry-1.docker.io/bitnamicharts tags: - bitnami-common + - name: openfga-operator + version: "0.1.0" + repository: "file://../openfga-operator" + condition: openfga-operator.enabled diff --git a/charts/openfga/README.md b/charts/openfga/README.md index a85e3815..1b04c622 100644 --- a/charts/openfga/README.md +++ b/charts/openfga/README.md @@ -151,6 +151,21 @@ datastore: passwordKey: password ``` +### Running migrations with the operator + +By default the chart runs database migrations from a Helm hook Job and gates the OpenFGA pods on it with an init container. Helm hooks are not run by Argo CD and conflict with `helm install --wait` and Flux, so the chart can instead install the [openfga-operator](../openfga-operator), which runs `openfga migrate` as a regular Job whenever the OpenFGA image or migration inputs change: + +```yaml +openfga-operator: + enabled: true + +datastore: + engine: postgres + uriSecret: my-postgres-secret +``` + +The operator only runs migrations; replicas, autoscaling and the pod template stay under the chart's control. It records the migrated version in the `-migration-status` ConfigMap and sets a `MigrationFailed` condition on the Deployment if a migration fails. The migration Job is built from the OpenFGA pod spec, so `sidecars` such as a database proxy and `extraInitContainers` run alongside it, and the `migrate.*` values (labels, non-hook annotations such as `sidecar.istio.io/inject: "false"`, extra volumes, init containers, sidecars, timeout) and `datastore.migrations.resources` are applied to it. The Job runs as the OpenFGA service account; set `migration.serviceAccount.create` to give it a dedicated `-migration` one, for example with cloud IAM annotations for DDL permissions. Migrations run when the image tag or the datastore connection settings change, so pin `image.tag` to a release rather than a floating tag; set `migration.trigger` to any new value to run one on demand. See the [operator README](../../operator/README.md) for how it works and its limitations. + ## Uninstalling the Chart To uninstall/delete the `openfga` deployment: diff --git a/charts/openfga/ci/operator-mode-values.yaml b/charts/openfga/ci/operator-mode-values.yaml new file mode 100644 index 00000000..f080f484 --- /dev/null +++ b/charts/openfga/ci/operator-mode-values.yaml @@ -0,0 +1,13 @@ +# Installs the openfga-operator subchart next to OpenFGA through chart-testing: +# subchart resolution, operator RBAC and the operator pod starting up. The +# memory datastore needs no migration, so the Deployment is not opted in to +# operator migrations; the Postgres migration path is covered by the operator +# E2E step in .github/workflows/test.yml. + +datastore: + engine: memory + +openfga-operator: + enabled: true + image: + pullPolicy: Never diff --git a/charts/openfga/templates/NOTES.txt b/charts/openfga/templates/NOTES.txt index 0048291e..8e7b13fc 100644 --- a/charts/openfga/templates/NOTES.txt +++ b/charts/openfga/templates/NOTES.txt @@ -1,3 +1,19 @@ +{{- if include "openfga.operatorMigrations" . }} +NOTE: database migrations are run by the openfga-operator. Whenever the OpenFGA +image or datastore settings change it runs the {{ include "openfga.fullname" . }}-migrate Job and +records the migrated version in the {{ include "openfga.fullname" . }}-migration-status ConfigMap. On a new database the +OpenFGA pods stay NotReady until the first migration completes. + +If the pods do not become ready, check the operator and the migration Job: + + kubectl logs -n {{ .Release.Namespace }} -l app.kubernetes.io/name=openfga-operator --tail=100 + kubectl get job/{{ include "openfga.fullname" . }}-migrate -n {{ .Release.Namespace }} -o yaml + kubectl describe deployment/{{ include "openfga.fullname" . }} -n {{ .Release.Namespace }} + +A `MigrationFailed` condition on the Deployment means the migration Job failed; +the operator retries it every 60s. + +{{ end -}} 1. Get the application URL by running these commands: {{- if .Values.ingress.enabled }} {{- range $host := .Values.ingress.hosts }} diff --git a/charts/openfga/templates/_helpers.tpl b/charts/openfga/templates/_helpers.tpl index 5889497a..312ee3a4 100644 --- a/charts/openfga/templates/_helpers.tpl +++ b/charts/openfga/templates/_helpers.tpl @@ -74,6 +74,52 @@ Create the name of the service account to use {{- end }} {{- end }} +{{/* +Create the name of the migration service account to use (operator mode only) +*/}} +{{- define "openfga.migrationServiceAccountName" -}} +{{- if .Values.migration.serviceAccount.create }} +{{- default (printf "%s-migration" (include "openfga.fullname" .)) .Values.migration.serviceAccount.name | trunc 63 | trimSuffix "-" }} +{{- else }} +{{- required "migration.serviceAccount.name must be set when migration.serviceAccount.create=false" .Values.migration.serviceAccount.name | trunc 63 | trimSuffix "-" }} +{{- end }} +{{- end }} + +{{/* +Identity of the datastore the operator migrates. The operator runs the migration +again whenever this changes, e.g. when the chart points at a different database +with the same OpenFGA image. Credentials in the URI are left out, so a password +change does not count and no secret goes into the hash. migration.trigger is any +user-chosen string that forces another run, for changes the chart cannot see +such as a Secret rotated under the same name. +*/}} +{{- define "openfga.migrationTrigger" -}} +{{- $ds := .Values.datastore -}} +{{- $uri := regexReplaceAll "^([a-z0-9+.-]+://)?[^@/]*@" (toString (default "" $ds.uri)) "${1}" -}} +{{- dict "engine" $ds.engine "uri" $uri "username" $ds.username "uriSecret" $ds.uriSecret "existingSecret" $ds.existingSecret "secretKeys" $ds.secretKeys "trigger" .Values.migration.trigger | toJson | sha256sum | trunc 16 -}} +{{- end -}} + +{{/* +migrate.annotations without Helm hook keys, as JSON for the operator to put on +the migration Job and its pod +*/}} +{{- define "openfga.migrationAnnotations" -}} +{{- $out := dict -}} +{{- range $k, $v := .Values.migrate.annotations -}} +{{- if not (hasPrefix "helm.sh/" $k) -}}{{- $_ := set $out $k (toString $v) -}}{{- end -}} +{{- end -}} +{{- toJson $out -}} +{{- end -}} + +{{/* +Return true if the openfga-operator runs the database migrations for this release +*/}} +{{- define "openfga.operatorMigrations" -}} +{{- if and (index .Values "openfga-operator" "enabled") .Values.datastore.applyMigrations (has .Values.datastore.engine (list "postgres" "mysql")) -}} +true +{{- end -}} +{{- end -}} + {{/* Return true if a secret object should be created */}} diff --git a/charts/openfga/templates/deployment.yaml b/charts/openfga/templates/deployment.yaml index 72bba418..26692a9c 100644 --- a/charts/openfga/templates/deployment.yaml +++ b/charts/openfga/templates/deployment.yaml @@ -4,9 +4,45 @@ metadata: name: {{ include "openfga.fullname" . }} labels: {{- include "openfga.labels" . | nindent 4 }} - {{- with .Values.annotations }} + {{- $operatorMigrations := include "openfga.operatorMigrations" . }} + {{- if or $operatorMigrations .Values.annotations }} annotations: + {{- if $operatorMigrations }} + openfga.dev/migration-enabled: "true" + openfga.dev/container-name: "{{ .Chart.Name }}" + openfga.dev/migration-trigger: {{ include "openfga.migrationTrigger" . | quote }} + {{- if or .Values.migration.serviceAccount.create .Values.migration.serviceAccount.name }} + openfga.dev/migration-service-account: '{{ include "openfga.migrationServiceAccountName" . }}' + {{- end }} + {{- with .Values.migrate.extraInitContainers }} + openfga.dev/migration-init-containers: {{ . | toJson | quote }} + {{- end }} + {{- with .Values.migrate.sidecars }} + openfga.dev/migration-sidecars: {{ include "common.tplvalues.render" (dict "value" . "context" $) | fromYamlArray | toJson | quote }} + {{- end }} + {{- with .Values.migrate.extraVolumes }} + openfga.dev/migration-volumes: {{ . | toJson | quote }} + {{- end }} + {{- with .Values.migrate.extraVolumeMounts }} + openfga.dev/migration-volume-mounts: {{ . | toJson | quote }} + {{- end }} + {{- with .Values.datastore.migrations.resources }} + openfga.dev/migration-resources: {{ . | toJson | quote }} + {{- end }} + {{- with .Values.migrate.timeout }} + openfga.dev/migration-timeout: {{ . | quote }} + {{- end }} + {{- with .Values.migrate.labels }} + openfga.dev/migration-labels: {{ toJson . | quote }} + {{- end }} + {{- $migrationAnnotations := include "openfga.migrationAnnotations" . }} + {{- if ne $migrationAnnotations "{}" }} + openfga.dev/migration-annotations: {{ $migrationAnnotations | quote }} + {{- end }} + {{- end }} + {{- with .Values.annotations }} {{- toYaml . | nindent 4 }} + {{- end }} {{- end }} spec: {{- if not .Values.autoscaling.enabled }} @@ -37,9 +73,10 @@ spec: serviceAccountName: {{ include "openfga.serviceAccountName" . }} securityContext: {{- toYaml .Values.podSecurityContext | nindent 8 }} - {{ if or (and (has .Values.datastore.engine (list "postgres" "mysql")) .Values.datastore.applyMigrations .Values.datastore.waitForMigrations) .Values.extraInitContainers }} + {{- $legacyMigrations := and (not (index .Values "openfga-operator" "enabled")) (has .Values.datastore.engine (list "postgres" "mysql")) }} + {{ if or (and $legacyMigrations .Values.datastore.applyMigrations .Values.datastore.waitForMigrations) .Values.extraInitContainers }} initContainers: - {{- if and (has .Values.datastore.engine (list "postgres" "mysql")) .Values.datastore.applyMigrations .Values.datastore.waitForMigrations (eq .Values.datastore.migrationType "job") }} + {{- if and $legacyMigrations .Values.datastore.applyMigrations .Values.datastore.waitForMigrations (eq .Values.datastore.migrationType "job") }} - name: wait-for-migration securityContext: {{- toYaml .Values.securityContext | nindent 12 }} @@ -49,7 +86,7 @@ spec: resources: {{- toYaml .Values.datastore.migrations.resources | nindent 12 }} {{- end }} - {{- if and (has .Values.datastore.engine (list "postgres" "mysql")) (eq .Values.datastore.migrationType "initContainer") }} + {{- if and $legacyMigrations (eq .Values.datastore.migrationType "initContainer") }} {{- with .Values.migrate.extraInitContainers }} {{- toYaml . | nindent 8 }} {{- end }} diff --git a/charts/openfga/templates/job.yaml b/charts/openfga/templates/job.yaml index a771fbea..05e3d749 100644 --- a/charts/openfga/templates/job.yaml +++ b/charts/openfga/templates/job.yaml @@ -1,4 +1,4 @@ -{{- if and (has .Values.datastore.engine (list "postgres" "mysql")) .Values.datastore.applyMigrations (eq .Values.datastore.migrationType "job") -}} +{{- if and (not (index .Values "openfga-operator" "enabled")) (has .Values.datastore.engine (list "postgres" "mysql")) .Values.datastore.applyMigrations (eq .Values.datastore.migrationType "job") -}} apiVersion: batch/v1 kind: Job metadata: diff --git a/charts/openfga/templates/rbac.yaml b/charts/openfga/templates/rbac.yaml index 3c8e0f8b..bbb5613d 100644 --- a/charts/openfga/templates/rbac.yaml +++ b/charts/openfga/templates/rbac.yaml @@ -1,4 +1,4 @@ -{{- if .Values.serviceAccount.create -}} +{{- if and (not (index .Values "openfga-operator" "enabled")) .Values.serviceAccount.create -}} apiVersion: rbac.authorization.k8s.io/v1 kind: Role metadata: diff --git a/charts/openfga/templates/serviceaccount.yaml b/charts/openfga/templates/serviceaccount.yaml index bbe191c9..bc3e4aea 100644 --- a/charts/openfga/templates/serviceaccount.yaml +++ b/charts/openfga/templates/serviceaccount.yaml @@ -10,3 +10,16 @@ metadata: {{- toYaml . | nindent 4 }} {{- end }} {{- end }} +{{- if and (include "openfga.operatorMigrations" .) .Values.migration.serviceAccount.create }} +--- +apiVersion: v1 +kind: ServiceAccount +metadata: + name: {{ include "openfga.migrationServiceAccountName" . }} + labels: + {{- include "openfga.labels" . | nindent 4 }} + {{- with .Values.migration.serviceAccount.annotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} +{{- end }} diff --git a/charts/openfga/tests/operator_mode_job_test.yaml b/charts/openfga/tests/operator_mode_job_test.yaml new file mode 100644 index 00000000..014e7abe --- /dev/null +++ b/charts/openfga/tests/operator_mode_job_test.yaml @@ -0,0 +1,27 @@ +suite: operator mode - job template +templates: + - templates/job.yaml +tests: + - it: should not render migration job when operator is enabled + set: + openfga-operator.enabled: true + datastore.engine: postgres + datastore.uri: "postgres://localhost/openfga" + datastore.applyMigrations: true + datastore.migrationType: job + asserts: + - hasDocuments: + count: 0 + + - it: should render migration job when operator is disabled + set: + openfga-operator.enabled: false + datastore.engine: postgres + datastore.uri: "postgres://localhost/openfga" + datastore.applyMigrations: true + datastore.migrationType: job + asserts: + - hasDocuments: + count: 1 + - isKind: + of: Job diff --git a/charts/openfga/tests/operator_mode_rbac_test.yaml b/charts/openfga/tests/operator_mode_rbac_test.yaml new file mode 100644 index 00000000..10ccba0a --- /dev/null +++ b/charts/openfga/tests/operator_mode_rbac_test.yaml @@ -0,0 +1,25 @@ +suite: operator mode - RBAC +templates: + - templates/rbac.yaml +tests: + - it: should not render legacy RBAC when operator is enabled + set: + openfga-operator.enabled: true + serviceAccount.create: true + asserts: + - hasDocuments: + count: 0 + + - it: should render legacy RBAC when operator is disabled + set: + openfga-operator.enabled: false + serviceAccount.create: true + asserts: + - hasDocuments: + count: 2 + - isKind: + of: Role + documentIndex: 0 + - isKind: + of: RoleBinding + documentIndex: 1 diff --git a/charts/openfga/tests/operator_mode_serviceaccount_test.yaml b/charts/openfga/tests/operator_mode_serviceaccount_test.yaml new file mode 100644 index 00000000..32959d61 --- /dev/null +++ b/charts/openfga/tests/operator_mode_serviceaccount_test.yaml @@ -0,0 +1,84 @@ +suite: operator mode - service accounts +templates: + - templates/serviceaccount.yaml +tests: + - it: should render migration service account when operator is enabled + set: + openfga-operator.enabled: true + datastore.engine: postgres + migration.serviceAccount.create: true + serviceAccount.create: true + asserts: + - hasDocuments: + count: 2 + - isKind: + of: ServiceAccount + documentIndex: 1 + - equal: + path: metadata.name + value: RELEASE-NAME-openfga-migration + documentIndex: 1 + + - it: should not render migration service account when operator is disabled + set: + openfga-operator.enabled: false + serviceAccount.create: true + asserts: + - hasDocuments: + count: 1 + + - it: should not render migration service account by default + set: + openfga-operator.enabled: true + datastore.engine: postgres + serviceAccount.create: true + asserts: + - hasDocuments: + count: 1 + + - it: should not render migration service account when migration SA creation is disabled + set: + openfga-operator.enabled: true + datastore.engine: postgres + migration.serviceAccount.create: false + migration.serviceAccount.name: external-sa + serviceAccount.create: true + asserts: + - hasDocuments: + count: 1 + + - it: should render migration service account with custom annotations + set: + openfga-operator.enabled: true + datastore.engine: postgres + migration.serviceAccount.create: true + migration.serviceAccount.annotations: + eks.amazonaws.com/role-arn: "arn:aws:iam::123456789012:role/openfga-migrator" + serviceAccount.create: true + asserts: + - equal: + path: metadata.annotations["eks.amazonaws.com/role-arn"] + value: "arn:aws:iam::123456789012:role/openfga-migrator" + documentIndex: 1 + + - it: should use custom migration service account name + set: + openfga-operator.enabled: true + datastore.engine: postgres + migration.serviceAccount.create: true + migration.serviceAccount.name: my-migrator + serviceAccount.create: true + asserts: + - equal: + path: metadata.name + value: my-migrator + documentIndex: 1 + + - it: should not render migration service account for the memory datastore + set: + openfga-operator.enabled: true + migration.serviceAccount.create: true + serviceAccount.create: true + asserts: + - hasDocuments: + count: 1 diff --git a/charts/openfga/tests/operator_mode_test.yaml b/charts/openfga/tests/operator_mode_test.yaml new file mode 100644 index 00000000..88a1dd94 --- /dev/null +++ b/charts/openfga/tests/operator_mode_test.yaml @@ -0,0 +1,282 @@ +suite: operator mode +templates: + - templates/deployment.yaml +tests: + # --- Deployment annotations --- + - it: should set operator annotations when operator and migration are enabled + set: + openfga-operator.enabled: true + datastore.engine: postgres + asserts: + - equal: + path: metadata.annotations["openfga.dev/migration-enabled"] + value: "true" + - equal: + path: metadata.annotations["openfga.dev/container-name"] + value: openfga + - isNull: + path: metadata.annotations["openfga.dev/migration-service-account"] + + - it: should set the migration service account annotation when the chart creates one + set: + openfga-operator.enabled: true + datastore.engine: postgres + migration.serviceAccount.create: true + asserts: + - equal: + path: metadata.annotations["openfga.dev/migration-service-account"] + value: RELEASE-NAME-openfga-migration + + - it: should pass migration Job configuration to the operator + set: + openfga-operator.enabled: true + datastore.engine: postgres + datastore.migrations.resources.requests.cpu: 100m + migrate.timeout: 2m + migrate.extraInitContainers: + - name: prepare-proxy + image: busybox:1.36 + migrate.sidecars: + - name: database-proxy + image: "{{ .Release.Name }}-proxy:v2" + migrate.extraVolumes: + - name: proxy-config + secret: + secretName: database-proxy + migrate.extraVolumeMounts: + - name: proxy-config + mountPath: /credentials + readOnly: true + migrate.annotations: + helm.sh/hook: post-install + admission.example.com/inject: enabled + migrate.labels: + app.kubernetes.io/component: overridden + network-policy.example.com/database: allowed + asserts: + - equal: + path: metadata.annotations["openfga.dev/migration-init-containers"] + value: '[{"image":"busybox:1.36","name":"prepare-proxy"}]' + - equal: + path: metadata.annotations["openfga.dev/migration-sidecars"] + value: '[{"image":"RELEASE-NAME-proxy:v2","name":"database-proxy"}]' + - equal: + path: metadata.annotations["openfga.dev/migration-volumes"] + value: '[{"name":"proxy-config","secret":{"secretName":"database-proxy"}}]' + - equal: + path: metadata.annotations["openfga.dev/migration-volume-mounts"] + value: '[{"mountPath":"/credentials","name":"proxy-config","readOnly":true}]' + - equal: + path: metadata.annotations["openfga.dev/migration-resources"] + value: '{"requests":{"cpu":"100m"}}' + - equal: + path: metadata.annotations["openfga.dev/migration-timeout"] + value: 2m + - equal: + path: metadata.annotations["openfga.dev/migration-annotations"] + value: '{"admission.example.com/inject":"enabled"}' + - equal: + path: metadata.annotations["openfga.dev/migration-labels"] + value: '{"app.kubernetes.io/component":"overridden","network-policy.example.com/database":"allowed"}' + + - it: should not set operator annotations when operator is disabled + set: + openfga-operator.enabled: false + annotations: + custom: value + asserts: + - isNull: + path: metadata.annotations["openfga.dev/migration-enabled"] + - equal: + path: metadata.annotations.custom + value: value + + - it: should not set operator annotations for the memory datastore + set: + openfga-operator.enabled: true + datastore.engine: memory + asserts: + - isNull: + path: metadata.annotations + + - it: should derive the migration trigger from the datastore settings + set: + openfga-operator.enabled: true + datastore.engine: postgres + datastore.uri: postgres://a/openfga + asserts: + - matchRegex: + path: metadata.annotations["openfga.dev/migration-trigger"] + pattern: ^[0-9a-f]{16}$ + + - it: should change the migration trigger when the database or migration.trigger changes + set: + openfga-operator.enabled: true + datastore.engine: postgres + datastore.uri: postgres://b/openfga + migration.trigger: "2" + asserts: + - notEqual: + path: metadata.annotations["openfga.dev/migration-trigger"] + # value for postgres://a/openfga with no explicit trigger, from the case above + value: 791636a2daefd486 + + - it: should not change the migration trigger when only the database password changes + set: + openfga-operator.enabled: true + datastore.engine: postgres + datastore.uri: postgres://openfga:other-password@a/openfga + asserts: + - equal: + path: metadata.annotations["openfga.dev/migration-trigger"] + value: 791636a2daefd486 + + - it: should forward migrate labels and non-hook annotations to the operator + set: + openfga-operator.enabled: true + datastore.engine: postgres + migrate.labels: + team: auth + migrate.annotations: + helm.sh/hook: post-install + sidecar.istio.io/inject: "false" + asserts: + - equal: + path: metadata.annotations["openfga.dev/migration-labels"] + value: '{"team":"auth"}' + - equal: + path: metadata.annotations["openfga.dev/migration-annotations"] + value: '{"sidecar.istio.io/inject":"false"}' + + - it: should not emit migration metadata annotations for hook-only annotations + set: + openfga-operator.enabled: true + datastore.engine: postgres + asserts: + - isNull: + path: metadata.annotations["openfga.dev/migration-labels"] + - isNull: + path: metadata.annotations["openfga.dev/migration-annotations"] + + - it: should use custom migration service account name when set + set: + openfga-operator.enabled: true + datastore.engine: postgres + migration.serviceAccount.name: my-custom-sa + asserts: + - equal: + path: metadata.annotations["openfga.dev/migration-service-account"] + value: my-custom-sa + + - it: should not set migration-service-account annotation by default + set: + openfga-operator.enabled: true + datastore.engine: postgres + asserts: + - isNull: + path: metadata.annotations["openfga.dev/migration-service-account"] + + # --- Replica count --- + # The operator never changes the replica count, so it renders as in legacy mode. + - it: should set replicas to replicaCount in operator mode + set: + openfga-operator.enabled: true + replicaCount: 3 + datastore.engine: postgres + asserts: + - equal: + path: spec.replicas + value: 3 + + - it: should leave replicas to the autoscaler in operator mode + set: + openfga-operator.enabled: true + autoscaling.enabled: true + datastore.engine: postgres + asserts: + - isNull: + path: spec.replicas + - equal: + path: metadata.annotations["openfga.dev/migration-enabled"] + value: "true" + + - it: should set replicas to replicaCount when operator is disabled + set: + openfga-operator.enabled: false + replicaCount: 5 + datastore.engine: postgres + asserts: + - equal: + path: spec.replicas + value: 5 + + # applyMigrations=false opts out: no operator annotations, replicas rendered normally. + - it: should not use operator mode when applyMigrations is false + set: + openfga-operator.enabled: true + replicaCount: 4 + datastore.engine: postgres + datastore.applyMigrations: false + asserts: + - isNull: + path: metadata.annotations["openfga.dev/migration-enabled"] + - equal: + path: spec.replicas + value: 4 + + # --- initContainers gating --- + - it: should not render migration initContainers when operator is enabled + set: + openfga-operator.enabled: true + datastore.engine: postgres + datastore.uri: "postgres://localhost/openfga" + datastore.applyMigrations: true + datastore.waitForMigrations: true + datastore.migrationType: job + asserts: + - isNull: + path: spec.template.spec.initContainers + + - it: should render migration initContainers when operator is disabled + set: + openfga-operator.enabled: false + datastore.engine: postgres + datastore.uri: "postgres://localhost/openfga" + datastore.applyMigrations: true + datastore.waitForMigrations: true + datastore.migrationType: job + asserts: + - isNotNull: + path: spec.template.spec.initContainers + + # --- Pod template labels --- + # The pod template must carry the full common label set (helm.sh/chart, + # component, version, managed-by, part-of) — not just selectorLabels — + # so logging/monitoring tooling that filters on these labels keeps working + # across upgrades. Regression guard for the operator-migration branch. + - it: should include common labels on pod template metadata when operator is disabled + set: + openfga-operator.enabled: false + asserts: + - isNotEmpty: + path: spec.template.metadata.labels["helm.sh/chart"] + - equal: + path: spec.template.metadata.labels["app.kubernetes.io/component"] + value: authorization-controller + - equal: + path: spec.template.metadata.labels["app.kubernetes.io/part-of"] + value: openfga + + - it: should include common labels on pod template metadata when operator is enabled + set: + openfga-operator.enabled: true + datastore.engine: postgres + asserts: + - isNotEmpty: + path: spec.template.metadata.labels["helm.sh/chart"] + - equal: + path: spec.template.metadata.labels["app.kubernetes.io/component"] + value: authorization-controller + - equal: + path: spec.template.metadata.labels["app.kubernetes.io/part-of"] + value: openfga diff --git a/charts/openfga/values.schema.json b/charts/openfga/values.schema.json index d1c296e9..e0348b90 100644 --- a/charts/openfga/values.schema.json +++ b/charts/openfga/values.schema.json @@ -1205,6 +1205,14 @@ }, "default": {} }, + "labels": { + "type": "object", + "description": "Map of labels to add to the migration job and pod", + "additionalProperties": { + "type": "string" + }, + "default": {} + }, "timeout": { "type": [ "string", @@ -1292,6 +1300,53 @@ "type": "boolean", "description": "This value is not used by this chart, but allows a common pattern of enabling/disabling subchart dependencies (where OpenFGA is a subchart)", "default": false + }, + "openfga-operator": { + "type": "object", + "description": "Configuration for the openfga-operator subchart, validated by that chart's own schema. When enabled, database migrations are run by the operator instead of the Helm hook Job.", + "properties": { + "enabled": { + "type": "boolean", + "description": "Install the openfga-operator and let it run database migrations", + "default": false + } + } + }, + "migration": { + "type": "object", + "description": "Settings for the migration Jobs the operator creates. Only used when openfga-operator.enabled is true.", + "properties": { + "trigger": { + "type": "string", + "description": "Change to any new value to run the migration again without changing the image or datastore settings", + "default": "" + }, + "serviceAccount": { + "type": "object", + "properties": { + "create": { + "type": "boolean", + "description": "Create a dedicated service account for migration Jobs; otherwise they run as the OpenFGA service account", + "default": false + }, + "annotations": { + "type": "object", + "description": "Annotations to add to the migration service account", + "additionalProperties": { + "type": "string" + }, + "default": {} + }, + "name": { + "type": "string", + "description": "The name of the migration service account. Defaults to {fullname}-migration. Must be set explicitly when create=false and a dedicated migration SA is desired; leave empty to skip the annotation entirely.", + "default": "" + } + }, + "additionalProperties": false + } + }, + "additionalProperties": false } }, "additionalProperties": false diff --git a/charts/openfga/values.yaml b/charts/openfga/values.yaml index 55bde0f2..8c40374c 100644 --- a/charts/openfga/values.yaml +++ b/charts/openfga/values.yaml @@ -372,10 +372,13 @@ migrate: extraVolumeMounts: [] extraInitContainers: [] sidecars: [] + # -- Added to the migration Job and its pod. With the operator, helm.sh/* keys + # are dropped and the rest is forwarded (e.g. sidecar.istio.io/inject: "false"). annotations: helm.sh/hook: "post-install, post-upgrade, post-rollback, post-delete" helm.sh/hook-weight: "-5" helm.sh/hook-delete-policy: "before-hook-creation" + # -- Added to the migration Job and its pod in both modes. labels: {} timeout: @@ -385,6 +388,44 @@ testContainerSpec: {} # -- Array of extra K8s manifests to deploy ## Note: Supports use of custom Helm templates extraObjects: [] + +# -- Configuration for the openfga-operator subchart. When enabled, database +# migrations are run by the operator instead of the Helm hook Job. +# See charts/openfga-operator/values.yaml for all available options. +openfga-operator: + enabled: false + # migrationJob: + # backoffLimit: 3 + # activeDeadlineSeconds: 0 + # ttlSecondsAfterFinished: 300 + # leaderElection: + # enabled: true + # watchNamespace: "" + # resources: + # requests: + # cpu: 10m + # memory: 64Mi + # limits: + # memory: 128Mi + +# -- Settings for the migration Jobs the operator creates. +# Only used when openfga-operator.enabled is true. +migration: + # -- The operator runs a migration when the OpenFGA image or the datastore + # settings change. Set this to any new value to run it again when neither + # changed, e.g. after rotating a Secret to point at a different database. + trigger: "" + serviceAccount: + # -- Create a dedicated service account for migration Jobs. By default the + # Job runs as the OpenFGA service account, like the Helm hook Job does. + create: false + # -- Annotations to add to the migration service account. + # Use this to attach cloud IAM roles (e.g., eks.amazonaws.com/role-arn) for DDL permissions. + annotations: {} + # -- The name of the migration service account. + # If not set and create is true, defaults to {fullname}-migration. + name: "" + ## Example: Deploy a PostgreSQL instance for dev/test using official Docker images. ## For production, use a managed database service or an operator like CloudnativePG. ## Configure the chart to use the secret: diff --git a/docs/adr/000-template.md b/docs/adr/000-template.md new file mode 100644 index 00000000..2cc78bb7 --- /dev/null +++ b/docs/adr/000-template.md @@ -0,0 +1,48 @@ +# ADR-NNN: Title + +- **Status:** Proposed +- **Date:** YYYY-MM-DD +- **Deciders:** [list of people involved] +- **Related Issues:** # +- **Related ADR:** [ADR-NNN](NNN-filename.md) + +## Context + +What is the problem or situation that motivates this decision? What constraints exist? What forces are at play? + +Include enough background that someone unfamiliar with the project can understand why this decision matters. + +## Decision + +What is the change being proposed or decided? + +### Alternatives Considered + +**A. [Alternative name]** + +[Description of the alternative] + +*Pros:* ... +*Cons:* ... + +**B. [Alternative name]** + +[Description of the alternative] + +*Pros:* ... +*Cons:* ... + +## Consequences + +### Positive + +- What improves as a result of this decision? + +### Negative + +- What gets harder, more complex, or more costly? + +### Risks + +- What assumptions might prove false? +- What could go wrong? diff --git a/docs/adr/001-adopt-openfga-operator.md b/docs/adr/001-adopt-openfga-operator.md new file mode 100644 index 00000000..c808a47c --- /dev/null +++ b/docs/adr/001-adopt-openfga-operator.md @@ -0,0 +1,117 @@ +# ADR-001: Adopt a Kubernetes Operator for OpenFGA Lifecycle Management + +- **Status:** Proposed +- **Date:** 2026-04-06 +- **Deciders:** OpenFGA Helm Charts maintainers +- **Related Issues:** #211, #107, #120, #100, #95, #126, #132, #144 + +## Context + +The OpenFGA Helm chart currently handles all lifecycle concerns — deployment, configuration, database migrations, and secret management — through Helm templates and hooks. This approach works for simple installations but breaks down in several important scenarios: + +1. **Database migrations rely on Helm hooks**, which are incompatible with GitOps tools (ArgoCD, FluxCD) and Helm's own `--wait` flag. This is the single biggest pain point for users, accounting for 6 open issues (#211, #107, #120, #100, #95, #126). + +2. **Store provisioning, authorization model updates, and tuple management** are runtime operations that happen through the OpenFGA API. There is no declarative, GitOps-native way to manage these. Teams must use imperative scripts, CI pipelines, or manual API calls to set up stores and push models after deployment. + +3. **The migration init container** depends on `groundnuty/k8s-wait-for`, an unmaintained image with known CVEs, pinned by mutable tag (#132, #144). + +4. **Migration and runtime workloads share a single ServiceAccount**, violating least-privilege when cloud IAM-based database authentication (AWS IRSA, GCP Workload Identity) maps the ServiceAccount directly to a database role (#95). + +### Alternatives Considered + +**A. Fix migrations within the Helm chart (no operator)** + +- Strip Helm hook annotations from the migration Job by default, rendering it as a regular resource. +- Replace `k8s-wait-for` with a shell-based init container that polls the database schema version directly. +- Add a separate ServiceAccount for the migration Job. + +*Pros:* Lower complexity, no new component to maintain. +*Cons:* Doesn't solve the ordering problem cleanly — the Job and Deployment are created simultaneously, requiring an init container to gate startup. Still requires an image or script to poll. Doesn't address store/model/tuple lifecycle at all. + +**B. Recommend initContainer mode as default** + +- Change `datastore.migrationType` default from `"job"` to `"initContainer"`, running migrations inside each pod. + +*Pros:* No separate Job, no hooks, no `k8s-wait-for`. +*Cons:* Every pod runs migrations on startup (wasteful). Rolling updates trigger redundant migrations. Crash-loops on migration failure. Still shares ServiceAccount. No path to store lifecycle management. + +**C. Build an operator (selected)** + +- A Kubernetes operator manages migrations as internal reconciliation logic and exposes CRDs for store, model, and tuple lifecycle. + +*Pros:* Solves all migration issues. Enables GitOps-native authorization management. Follows established Kubernetes patterns (CNPG, Strimzi, cert-manager). Separates concerns cleanly. +*Cons:* Significant development and maintenance investment. New component to deploy and monitor. Learning curve for contributors. + +**D. External migration tool (e.g., Flyway, golang-migrate)** + +- Remove migrations from the chart entirely and document using an external tool. + +*Pros:* Simplifies the chart completely. +*Cons:* Shifts complexity to the user. Every user must build their own migration pipeline. No standard approach across the community. + +## Decision + +We will build an **OpenFGA Kubernetes Operator**. This ADR decides Stage 1 only: + +1. **Database migration orchestration** (Stage 1) — replacing Helm hooks, the `k8s-wait-for` init container, and shared ServiceAccount with operator-managed migration Jobs. + +2. **Declarative store lifecycle management** (Stages 2-4) — `FGAStore`, `FGAModel`, and `FGATuples` CRDs for GitOps-native authorization configuration. Under consideration; each stage needs its own ADR before implementation. + +The operator will be: +- Written in Go using `controller-runtime` / kubebuilder +- Distributed as a Helm subchart dependency of the main OpenFGA chart +- Optional — `openfga-operator.enabled` defaults to `false`, which keeps the existing behavior + +Development will follow a staged approach to deliver value incrementally: + +| Stage | Scope | Outcome | +|-------|-------|---------| +| 1 | Operator scaffolding + migration handling | All 6 migration issues resolved | +| 2 | `FGAStore` CRD | Declarative store provisioning | +| 3 | `FGAModel` CRD | Declarative authorization model management | +| 4 | `FGATuples` CRD | Declarative tuple management | + +## Implementation Status + +Stage 1 is implemented alongside this ADR (openfga chart 0.4.0, openfga-operator chart 0.1.0). Stages 2-4 are not implemented. + +### Delivered in Stage 1 + +- Operator Go project under `/operator/`, built with `controller-runtime` +- Operator packaged as a Helm subchart (`charts/openfga-operator/`) and wired into the main chart via a `condition: openfga-operator.enabled` dependency +- `openfga-operator.enabled` values toggle (default `false`) that gates all operator-managed behavior +- Migration reconciler (`migration_controller.go`) that runs migration Jobs when the operator is enabled +- Optional separate migration ServiceAccount with IAM-annotation support (`openfga.migrationServiceAccountName` helper, `migration.serviceAccount.create`) + +### Deferred to later stages + +- `FGAStore`, `FGAModel`, and `FGATuples` CRDs and their controllers +- Declarative store/model/tuple lifecycle management + +### Backward-compatibility path + +When `openfga-operator.enabled: false` (the default), the chart still renders the legacy migration path: the Helm-hook migration Job, the `groundnuty/k8s-wait-for` init container, and the job-status RBAC. Making the operator the default and retiring this path is left to a later ADR. + +## Consequences + +### Positive + +- **Resolves the migration issues** (#211, #107, #120, #100, #126) and related dependency issues (#132, #144) on the operator-enabled path; #95 is addressed by the opt-in migration ServiceAccount +- **Removes `k8s-wait-for` from the operator-enabled path** — the unmaintained, CVE-carrying image is no longer used when `openfga-operator.enabled: true`, and would leave the chart entirely if the legacy path is retired +- **Enables GitOps-native authorization management** (planned, Stages 2-4) — stores, models, and tuples will become declarative Kubernetes resources that ArgoCD/FluxCD can sync +- **Enables least-privilege** — `migration.serviceAccount.create` gives migration Jobs (DDL) a ServiceAccount separate from the runtime (CRUD) on the operator-enabled path +- **Path to simplifying the Helm chart** — the migration Job template, init container logic, job-status RBAC, and hook annotations are conditionalized behind `openfga-operator.enabled: false` and could be removed if the legacy path is retired +- **Follows Kubernetes ecosystem conventions** — operators are the standard pattern for managing stateful application lifecycle + +### Negative + +- **New component to maintain** — the operator is a full Go project with its own release cycle, CI, testing, and CVE surface +- **Increased deployment footprint** — an additional pod running in the cluster (default requests 10m CPU, 64Mi memory) +- **Learning curve** — contributors need to understand controller-runtime patterns to modify the operator +- **CRD management complexity** (applies once Stages 2-4 land) — Helm does not upgrade or delete CRDs; users may need to apply CRD manifests separately on operator upgrades +- **Two code paths** — the chart must maintain both the operator-enabled path and the legacy path + +### Neutral + +- **Backward compatibility preserved** — `openfga-operator.enabled: false` (the default) keeps the existing Helm-hook behavior unchanged +- **No change for memory-datastore users** — users running with `datastore.engine: memory` are unaffected (no migrations, no operator needed) diff --git a/docs/adr/002-operator-managed-migrations.md b/docs/adr/002-operator-managed-migrations.md new file mode 100644 index 00000000..baa5a015 --- /dev/null +++ b/docs/adr/002-operator-managed-migrations.md @@ -0,0 +1,238 @@ +# ADR-002: Replace Helm Hook Migrations with Operator-Managed Migrations + +- **Status:** Proposed +- **Date:** 2026-04-06 +- **Deciders:** OpenFGA Helm Charts maintainers +- **Related ADR:** [ADR-001](001-adopt-openfga-operator.md) +- **Related Issues:** #211, #107, #120, #100, #95, #126, #132, #144 + +## Context + +### How Migrations Work Today + +The current Helm chart uses a **Helm hook Job** to run database migrations (`openfga migrate`) and a **`k8s-wait-for` init container** on the Deployment to block server startup until the migration completes. + +Seven files are involved: + +| File | Role | +|------|------| +| `templates/job.yaml` | Migration Job with Helm hook annotations | +| `templates/deployment.yaml` | OpenFGA Deployment + `wait-for-migration` init container | +| `templates/serviceaccount.yaml` | Shared ServiceAccount (migration + runtime) | +| `templates/rbac.yaml` | Role + RoleBinding so init container can poll Job status | +| `templates/_helpers.tpl` | Datastore environment variable helpers | +| `values.yaml` | `datastore.*`, `migrate.*`, `initContainer.*` configuration | +| `Chart.yaml` | `bitnami/common` dependency for migration sidecars | + +**The migration Job** (`templates/job.yaml`) is annotated as a Helm hook: + +```yaml +annotations: + "helm.sh/hook": post-install,post-upgrade,post-rollback,post-delete + "helm.sh/hook-delete-policy": before-hook-creation + "helm.sh/hook-weight": "1" +``` + +This means Helm manages it outside the normal release lifecycle — it only runs after Helm finishes creating/upgrading all other resources. + +**The wait-for init container** blocks the Deployment pods from starting: + +```yaml +initContainers: + - name: wait-for-migration + image: "groundnuty/k8s-wait-for:v2.0" + args: ["job-wr", "openfga-migrate"] +``` + +It polls the Kubernetes API (`GET /apis/batch/v1/.../jobs/openfga-migrate`) until `.status.succeeded >= 1`. This requires RBAC permissions (Role/RoleBinding for `batch/jobs` `get`/`list`). + +**The alternative mode** (`datastore.migrationType: initContainer`) runs migration directly inside each Deployment pod as an init container, avoiding hooks entirely but introducing redundant migration runs across replicas. + +### The Six Issues + +| Issue | Tool | Root Cause | +|-------|------|-----------| +| **#211** | ArgoCD | ArgoCD ignores Helm hook annotations. The migration Job is never created as a managed resource. The init container waits forever for a Job that doesn't exist. | +| **#107** | ArgoCD | Same root cause. The Job is invisible in ArgoCD's UI — users can't see, debug, or manually sync it. | +| **#120** | Helm `--wait` | Circular deadlock. Helm waits for the Deployment to be ready before running post-install hooks. The Deployment is never ready because the init container waits for the hook Job. The Job never runs because Helm is waiting. | +| **#100** | FluxCD | FluxCD waits for all resources by default. The `hook-delete-policy: before-hook-creation` removes the completed Job before FluxCD can confirm the Deployment is healthy. | +| **#95** | AWS IRSA | Migration and runtime share a ServiceAccount. With IAM-based DB auth, the runtime gets DDL permissions it doesn't need (CREATE TABLE, ALTER TABLE). | +| **#126** | All | The `k8s-wait-for` image is configured in two separate places in `values.yaml`, leading to inconsistency. Related: #132 (image unmaintained, has CVEs) and #144 (pinned by mutable tag). | + +### Why Helm Hooks Are Fundamentally Wrong for This + +Helm hooks are a **deploy-time orchestration mechanism**. They assume Helm is the active agent running the deployment. GitOps tools (ArgoCD, FluxCD) break this assumption — they render the chart to manifests and apply them declaratively. The hook annotations are either ignored (ArgoCD) or cause ordering/cleanup conflicts (FluxCD). + +This is not a bug in ArgoCD or FluxCD. It is a fundamental mismatch between Helm's imperative hook model and the declarative GitOps model. + +## Decision + +Replace the Helm hook migration Job and `k8s-wait-for` init container with **operator-managed migrations** as part of Stage 1 of the OpenFGA Operator (see [ADR-001](001-adopt-openfga-operator.md)). + +### How It Works + +The operator runs a **migration controller** that reconciles the OpenFGA Deployment: + +```text +┌──────────────────────────────────────────────────────────┐ +│ Operator Reconciliation │ +│ │ +│ 1. Read Deployment and derive migration identity │ +│ 2. Read ConfigMap/openfga-migration-status │ +│ └── "Last migrated image and datastore trigger" │ +│ 3. Identities differ → migration needed │ +│ 4. Create Job/openfga-migrate │ +│ ├── ServiceAccount: openfga (or a dedicated one) │ +│ ├── Image: openfga/openfga:v1.14.0 │ +│ ├── Args: ["migrate"] │ +│ └── ttlSecondsAfterFinished: 300 │ +│ 5. Watch Job until succeeded │ +│ 6. Update ConfigMap → "version: v1.14.0" │ +└──────────────────────────────────────────────────────────┘ +``` + +**Key design decisions within this approach:** + +#### The operator only runs migrations + +The operator creates Jobs and records their outcome; it never changes the Deployment's replica count or pod template. The chart renders `spec.replicas` exactly as in legacy mode (or leaves it to an HPA), so `kubectl scale`, autoscalers and GitOps tools behave the same whether or not the operator is enabled. + +Readiness comes from OpenFGA itself: `IsReady()` reports `NOT_SERVING` while the schema revision is below `MinimumSupportedDatastoreSchemaRevision` (4 since v1.3.x). On a **fresh install** the database is empty, so every pod stays `NotReady` until the first migration Job completes, and `helm install --wait` returns once it has. On an **upgrade** the existing schema already meets that minimum, so new pods pass readiness right away and serve on the previous schema while the Job applies the newer migrations. This relies on OpenFGA migrations being backward compatible, which is also what the Helm hook flow has always done: its init container sees the previous release's completed hook Job and lets new pods start before the new hook runs. + +**Rejected alternative — let the operator own the replica count:** the chart could omit `spec.replicas` (or render 0) and have the operator scale the Deployment up once the migration succeeds. Testing this showed three problems: switching an existing release to operator mode removes the field, so both Helm's three-way merge and server-side apply reset the Deployment to one replica until the migration finishes; `kubectl scale` and HPAs are overridden by the operator; and the scale-up buys nothing on upgrades, where the readiness check does not hold pods back. + +#### Migration identity tracking via ConfigMap + +A ConfigMap (`openfga-migration-status`) records the last successfully migrated identity: the image version and the datastore trigger the chart derives from the connection settings (without credentials). The operator compares this to the Deployment to determine if migration is needed. This is: +- Simple to inspect (`kubectl get configmap openfga-migration-status -o yaml`) +- Survives operator restarts +- Can be manually deleted to force re-migration once the previous migration Job has been cleaned up + +#### Separate ServiceAccount for migrations + +Migration Jobs run as the OpenFGA ServiceAccount by default, as the Helm hook Job does. With `migration.serviceAccount.create` the chart creates a dedicated `{fullname}-migration` ServiceAccount instead, which can carry cloud IAM roles that grant DDL permissions while the runtime ServiceAccount keeps only CRUD permissions. + +#### Migration Job is a regular resource + +The Job created by the operator has no Helm hook annotations. It is a standard Kubernetes Job, visible to ArgoCD, FluxCD, and all Kubernetes tooling. It has an owner reference to the OpenFGA Deployment, so it is garbage collected with it. + +#### Failure handling + +| Failure | Behavior | +|---------|----------| +| Job fails | Operator sets `MigrationFailed` on the Deployment, keeps the failed Job for 60 seconds so its logs can be read, then replaces it. On a fresh database the pods stay `NotReady`; on an upgrade they keep serving on the previous schema. | +| Job pod never starts | A bad secret reference, image pull error or unschedulable pod never fails the Job. Once the Deployment's pod template changes (the fix rolls out), the operator rebuilds a Job whose pod is not running. | +| Image changes while a Job runs | The running Job is left to finish and then replaced by one for the new image. The hook flow deletes the running hook Job instead (`before-hook-creation`), which can abort a concurrent index build and leave it invalid. Replacement uses foreground deletion, so the next Job only starts once the old pods are gone. | +| Datastore repointed, same image | The chart derives a trigger from the datastore settings that is part of the migration identity, so the migration runs against the new database. `migration.trigger` forces a run for changes the chart cannot see. | +| Same-name Job or ConfigMap from elsewhere | Not touched; a `MigrationJobConflict` event is recorded until it is removed. | +| Job hangs | No deadline by default, like the Helm hook Job. `activeDeadlineSeconds` can be set, but a migration cut off halfway (an index build, a MySQL table rebuild) starts over on the next attempt. | +| Operator crashes | On restart, re-reads the ConfigMap and Job status and resumes. The retry delay is measured from the failed Job's condition, so it survives restarts. | +| Database unreachable | Job fails to connect. After exhausting `backoffLimit` the cycle above repeats until the database becomes available. | + +### Sequence Comparison + +**Before (Helm hooks):** + +```text +helm install + ├── Create ServiceAccount, RBAC, Secret, Service + ├── Create Deployment (with wait-for-migration init container) + │ └── Pod starts → init container polls for Job → waits... + ├── [Helm finishes regular resources] + ├── Run post-install hooks: + │ └── Create Job/openfga-migrate → runs openfga migrate + │ └── Job succeeds + ├── Init container sees Job succeeded → exits + └── Main container starts +``` + +Problems: ArgoCD skips step 4. FluxCD deletes Job in step 4. `--wait` deadlocks between steps 2 and 4. + +**After (operator-managed, fresh install):** + +```text +helm install + ├── Create ServiceAccount (plus a migration ServiceAccount if enabled) + ├── Create Secret, Service + ├── Create Deployment (no init containers) + ├── Create Operator Deployment + └── [Helm is done — all resources are regular, no hooks] + +(The OpenFGA pods start but stay NotReady: the database has no schema yet.) + +Operator starts: + ├── Detects Deployment image version + ├── No migration status ConfigMap → migration needed + ├── Creates Job/openfga-migrate (regular Job, no hooks) + │ └── Runs as the OpenFGA ServiceAccount (or the migration one) + │ └── Runs openfga migrate → succeeds + ├── Creates ConfigMap with migrated version + └── Pods pass readiness +``` + +**After (operator-managed, upgrade with new image):** + +```text +helm upgrade + ├── Patches Deployment with new image tag + ├── Kubernetes starts rolling update + │ └── New pods (v1.14) pass readiness on the previous schema + └── [Helm is done] + +Operator reconciles: + ├── Detects image version differs from ConfigMap + ├── Creates Job/openfga-migrate → runs migration + └── Updates ConfigMap → "version: v1.14.0" +``` + +No Helm hooks. No migration waiter in the OpenFGA Deployment. No `k8s-wait-for`. All resources are regular Kubernetes objects. + +### What Changes in the Helm Chart + +Nothing is deleted outright — every change is gated on `openfga-operator.enabled` so the legacy flow remains the default for backward compatibility. + +**Gated on `openfga-operator.enabled: false` (legacy Helm-hook flow, rendered when the operator is disabled):** + +| File/Section | Behavior when operator is enabled | +|--------------|-----------------------------------| +| `templates/job.yaml` | Skipped — operator creates migration Jobs dynamically | +| `templates/rbac.yaml` | Skipped — no init container needs to poll Job status | +| `values.yaml`: `initContainer.*` | Unused — `k8s-wait-for` not deployed | +| `values.yaml`: `datastore.migrationType`, `datastore.waitForMigrations` | Unused — operator always uses a Job and handles ordering | +| Deployment migration init containers | Skipped — OpenFGA's readiness check holds pods until the schema is migrated | + +**Added (active only when `openfga-operator.enabled: true`):** + +| File/Section | Purpose | +|--------------|---------| +| `values.yaml`: `openfga-operator.enabled` | Toggle the operator subchart | +| `values.yaml`: `openfga-operator.migrationJob.*` | Migration Job backoff, deadline, and TTL configuration | +| `values.yaml`: `migration.serviceAccount.*` | Optional separate ServiceAccount for migration Jobs | +| `values.yaml`: `migration.trigger` | Explicit rerun trigger for referenced Secret data changes | +| `values.yaml`: migration pod values | `migrate.extraInitContainers`, `migrate.sidecars`, volumes, mounts, resources, timeout, non-hook annotations, and labels are forwarded to operator Jobs | +| `templates/serviceaccount.yaml`: second SA | Optional migration ServiceAccount | +| `charts/openfga-operator/` | Operator subchart (conditional dependency) | + +Users on `openfga-operator.enabled: false` (the default) see identical rendered output to the pre-operator chart, so gradual adoption is possible with no forced migration. + +## Consequences + +### Positive + +- **Migration issues resolved** (#211, #107, #120, #100, #126) — no Helm hooks means no ArgoCD/FluxCD/`--wait` incompatibility +- **`k8s-wait-for` eliminated** — removes an unmaintained image with CVEs from the supply chain (#132, #144) +- **Least-privilege available** — `migration.serviceAccount.create` separates the migration (DDL) and runtime (CRUD) ServiceAccounts (#95) +- **Runtime surface area reduced** — when `openfga-operator.enabled: true`, the legacy migration Job, init-container `k8s-wait-for` logic, and job-watching RBAC are skipped from the rendered manifest +- **Migration is observable** — Job is a regular resource visible in all tools; ConfigMap records migration history; operator conditions surface errors +- **Idempotent and crash-safe** — operator can restart at any point and resume correctly + +### Negative + +- **Operator is a new runtime dependency** — if the operator pod is unavailable, migrations don't run (but existing running pods are unaffected) +- **Two upgrade paths to document** — `openfga-operator.enabled: true` (new) vs `openfga-operator.enabled: false` (legacy) + +### Risks + +- **Readiness relies on OpenFGA's schema check** — pods on a fresh database are held back only by `MinimumSupportedDatastoreSchemaRevision` in `pkg/storage/sqlcommon/sqlcommon.go`, and upgrades rely on each release working against the previous schema. Both are OpenFGA guarantees the Helm hook flow already depended on. +- **Migrations run as soon as their inputs change:** As with the hook Job, nothing drains traffic first. Some migrations, such as MySQL's `008_collate_identifiers` in v1.18.0, block writes while tables are rebuilt; OpenFGA's runbook recommends draining traffic for those, which stays a manual step. +- **ConfigMap as state store:** If the ConfigMap is accidentally deleted, the operator records the migration identity again from the completed Job while it exists, or reruns the migration once the Job has been cleaned up. `openfga migrate` is idempotent. diff --git a/docs/adr/README.md b/docs/adr/README.md new file mode 100644 index 00000000..d6b3445e --- /dev/null +++ b/docs/adr/README.md @@ -0,0 +1,178 @@ +# Architecture Decision Records + +This directory contains Architecture Decision Records (ADRs) for the OpenFGA Helm Charts project. + +ADRs are short documents that capture significant architectural decisions along with their context, alternatives considered, and consequences. They serve as a decision log — not a living design doc, but a point-in-time record of *why* a decision was made. + +We follow the format described by [Michael Nygard](https://cognitect.com/blog/2011/11/15/documenting-architecture-decisions). + +## Index + +| ADR | Title | Status | Date | +|-----|-------|--------|------| +| [ADR-001](001-adopt-openfga-operator.md) | Adopt a Kubernetes Operator for OpenFGA Lifecycle Management | Proposed | 2026-04-06 | +| [ADR-002](002-operator-managed-migrations.md) | Replace Helm Hook Migrations with Operator-Managed Migrations | Proposed | 2026-04-06 | + +--- + +## What is an ADR? + +An ADR captures a single architectural decision. It records: + +- **What** was decided +- **Why** it was decided (the context and constraints at the time) +- **What alternatives** were considered and why they were rejected +- **What consequences** follow from the decision (positive, negative, and neutral) + +ADRs are **immutable once accepted** — if a decision changes, you write a new ADR that supersedes the old one rather than editing it. This preserves the history of *why* things changed over time. + +## ADR Lifecycle + +```text +Proposed → Accepted → (optionally) Superseded or Deprecated + ↑ + │ feedback loop + │ + Discussion +``` + +### Statuses + +| Status | Meaning | +|--------|---------| +| **Proposed** | The ADR has been written and is open for discussion. No commitment has been made. | +| **Accepted** | The decision has been agreed upon by maintainers. Implementation can proceed. | +| **Deprecated** | The decision is no longer relevant (e.g., the feature was removed). | +| **Superseded by ADR-XXX** | A newer ADR has replaced this decision. The old ADR links to the new one. | + +## How to Propose an ADR + +1. **Create a branch** — e.g., `docs/adr-005-my-decision` + +2. **Copy the template** — use `000-template.md` as a starting point + +3. **Write the ADR** — fill in Context, Decision, and Consequences. Focus on *why*, not *how*. The most valuable part is the Alternatives Considered section — it shows reviewers what you evaluated and why you chose this path. + +4. **Assign a number** — use the next sequential number. Check the index above. + +5. **Open a pull request** — the PR is where discussion happens. Title it: `ADR-005: ` + +6. **Add to the index** — update the table in this README with the new entry (status: Proposed) + +### Proposing related ADRs together + +When multiple ADRs are part of a single cohesive proposal — e.g., a foundational decision and several downstream decisions that depend on it — they can be submitted in a single PR. This lets reviewers see the full picture instead of bouncing between separate PRs. + +When doing this: + +- **Explain the relationship in the PR description** — identify which ADR is the foundational decision and which are downstream. For example: "ADR-001 is the core decision to build an operator. ADR-002 is a downstream decision about how the operator handles migrations." +- **Each ADR can be accepted or rejected independently** — a reviewer might approve the foundational decision but push back on a downstream one. If that happens, split the PR: merge the accepted ADRs and keep the contested ones open for further discussion. +- **Keep each ADR self-contained** — even though they're in the same PR, each ADR should stand on its own. A reader should be able to understand a downstream ADR without reading the foundational one first (though they may reference each other). + +## How to Give Feedback on an ADR + +ADR review happens in the **pull request**, not by editing the ADR directly. This keeps the discussion visible and linked to the decision. + +### As a reviewer + +- **Comment on the PR** — ask questions, challenge assumptions, suggest alternatives. Good review questions: + - "Did you consider X as an alternative?" + - "What happens if Y fails?" + - "This conflicts with how we do Z — can you address that?" + - "I agree with the decision but the consequence about X should mention Y" + +- **Request changes** if you believe the decision is wrong or incomplete + +- **Approve** when you're satisfied the decision is sound and well-documented + +### As the author responding to feedback + +- **Update the ADR in the PR** based on feedback: + - Add alternatives that reviewers suggested (with your evaluation of them) + - Expand the Consequences section if reviewers identified impacts you missed + - Clarify the Context if reviewers were confused about the problem + - Adjust the Decision if feedback reveals a better approach + +- **Do NOT delete feedback-driven changes** — if a reviewer raised a valid alternative and you addressed it, the ADR is stronger for including it + +- **Resolve PR comments** as you address them so reviewers can track progress + +### Reaching consensus + +- ADRs move to **Accepted** when maintainers approve the PR +- Not every maintainer needs to approve — follow the project's normal review standards +- If consensus can't be reached, escalate to a synchronous discussion (meeting, call) and record the outcome in the PR +- Disagreement is fine — document it in the Consequences section as a risk or trade-off rather than hiding it + +## How to Supersede an ADR + +When a decision needs to change: + +1. **Do NOT edit the original ADR** — it's a historical record + +2. **Write a new ADR** that references the old one: + ```markdown + - **Supersedes:** [ADR-002](002-operator-managed-migrations.md) + ``` + +3. **Update the old ADR's status** — change it to: + ```markdown + - **Status:** Superseded by [ADR-007](007-new-approach.md) + ``` + +4. **Update the index** in this README + +This way, anyone reading ADR-002 knows it's been replaced and can follow the link to understand what changed and why. + +## ADR Format + +Every ADR follows this structure: + +```markdown +# ADR-NNN: Title + +- **Status:** Proposed | Accepted | Deprecated | Superseded by ADR-XXX +- **Date:** YYYY-MM-DD +- **Deciders:** Who was involved in the decision +- **Related Issues:** GitHub issue references +- **Related ADR:** Links to related ADRs + +## Context + +What is the problem or situation that motivates this decision? +Include enough background that someone unfamiliar with the project +can understand why this decision matters. + +## Decision + +What is the decision and why was it chosen? + +### Alternatives Considered + +What other options were evaluated? Why were they rejected? +This is often the most valuable section — it prevents future +contributors from re-proposing rejected approaches. + +## Consequences + +### Positive +What improves as a result of this decision? + +### Negative +What gets harder or more complex? Be honest — every decision has costs. + +### Risks +What could go wrong? What assumptions might prove false? +``` + +## Template + +A blank template is available at [000-template.md](000-template.md). + +## Tips for Writing Good ADRs + +- **Keep it short** — an ADR is one decision, not a design doc. If it's longer than 2-3 pages, consider splitting it. +- **Focus on why, not how** — implementation details change; the reasoning behind the decision is what matters long-term. +- **Be honest about trade-offs** — an ADR that lists only positive consequences isn't credible. Every decision has costs. +- **Write for your future self** — in 18 months, you won't remember why you chose this. The ADR should tell you. +- **Not every decision needs an ADR** — use ADRs for decisions that are hard to reverse, affect multiple components, or where the reasoning isn't obvious from the code. diff --git a/operator/.dockerignore b/operator/.dockerignore new file mode 100644 index 00000000..3efb8a0e --- /dev/null +++ b/operator/.dockerignore @@ -0,0 +1,6 @@ +**/.git +**/.gitignore +**/README.md +**/LICENSE +**/Makefile +**/.dockerignore diff --git a/operator/Dockerfile b/operator/Dockerfile new file mode 100644 index 00000000..1d1d71c4 --- /dev/null +++ b/operator/Dockerfile @@ -0,0 +1,25 @@ +# pinned multi-arch index for golang:1.26.8 (linux/amd64, linux/arm64, ...) +FROM --platform=$BUILDPLATFORM golang:1.26.8@sha256:6c2a5538f964f1c82f97ad14988bf05de100d922d159d0e398b54c7b0ca0c6c9 AS builder + +# buildx provides these automatically; declare so Go cross-compiles to the +# requested target instead of the build host's arch. +ARG TARGETOS +ARG TARGETARCH + +WORKDIR /workspace +COPY go.mod go.sum ./ +RUN go mod download + +COPY cmd/ cmd/ +COPY internal/ internal/ + +RUN CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} \ + go build -ldflags="-s -w" -o /operator ./cmd/ + +# pinned multi-arch index for gcr.io/distroless/static:nonroot +FROM gcr.io/distroless/static:nonroot@sha256:e3f945647ffb95b5839c07038d64f9811adf17308b9121d8a2b87b6a22a80a39 +WORKDIR / +COPY --from=builder /operator . +USER 65532:65532 + +ENTRYPOINT ["/operator"] diff --git a/operator/Makefile b/operator/Makefile new file mode 100644 index 00000000..4b97c0cb --- /dev/null +++ b/operator/Makefile @@ -0,0 +1,27 @@ +IMG ?= openfga/openfga-operator:dev + +.PHONY: build test vet fmt lint docker-build docker-push clean + +build: + mkdir -p bin + go build -o bin/operator ./cmd/ + +test: + go test ./... -v + +vet: + go vet ./... + +fmt: + go fmt ./... + +lint: vet fmt + +docker-build: + docker build -t $(IMG) . + +docker-push: + docker push $(IMG) + +clean: + rm -rf bin/ diff --git a/operator/README.md b/operator/README.md new file mode 100644 index 00000000..e9e6ff49 --- /dev/null +++ b/operator/README.md @@ -0,0 +1,187 @@ +# OpenFGA Operator + +A Kubernetes operator that manages database migrations for OpenFGA deployments. Instead of relying on Helm hooks and init containers, the operator watches OpenFGA Deployments, detects image and datastore changes, and orchestrates migrations as regular Jobs. + +This is **Stage 1** of the operator — focused solely on migration orchestration. See [ADR-001](../docs/adr/001-adopt-openfga-operator.md) for the full roadmap. + +## How It Works + +1. The operator watches Deployments in its configured namespace, which defaults to the operator pod's namespace, labeled `app.kubernetes.io/part-of: openfga` and `app.kubernetes.io/component: authorization-controller` +2. When the migration identity changes (the container image tag plus the `openfga.dev/migration-trigger` annotation, compared to the `{name}-migration-status` ConfigMap), the operator: + - Creates a migration Job running `openfga migrate`, using the Deployment's image, environment, pod scheduling, init containers, and other containers as [native sidecars](https://kubernetes.io/docs/concepts/workloads/pods/sidecar-containers/) + - Applies migration-specific containers, volumes, mounts, resources, timeout, labels, and annotations from the Deployment annotations rendered by the chart + - Waits for the Job to complete + - Records the identity in the ConfigMap and applies `ttlSecondsAfterFinished` so Kubernetes cleans the Job up +3. On failure, a `MigrationFailed` condition is set on the Deployment. The failed Job is kept for 60 seconds so its logs can be inspected, then replaced with a new one. + +A running migration is never interrupted. If the image changes again while a Job's pod is running (a rollback, or two upgrades in a row), the operator waits for that Job to finish and then runs the migration for the new image. Aborting a non-transactional step such as Postgres's concurrent index build in migration 006 leaves an invalid index that the next run skips. To abort a migration that is stuck, delete the Job or set `migrationJob.activeDeadlineSeconds`. + +The operator never changes the Deployment's replica count or pod template. On a new database, OpenFGA's readiness check (`MinimumSupportedDatastoreSchemaRevision`) keeps pods `NotReady` until the first migration has run. On an upgrade the existing schema already meets that minimum, so new pods serve on it while the Job applies the newer migrations, which is the same behaviour as the Helm hook flow. + +## Prerequisites + +- Go 1.26.8+ +- Docker +- Helm 3.6+ +- A Kubernetes cluster (Rancher Desktop, kind, etc.), 1.29 or newer when the OpenFGA pod has sidecars + +## Development + +### Build + +```bash +cd operator +go build ./... +``` + +### Test + +```bash +go test ./... -v +``` + +### Lint + +```bash +go vet ./... +``` + +### Docker Image + +```bash +docker build -t openfga/openfga-operator:dev . +``` + +## Releasing + +CI publishes `ghcr.io/openfga/openfga-operator:<appVersion>` on the first push to `main` that carries that appVersion and never overwrites it, and chart-releaser likewise skips chart versions that already exist. A change to the operator image (`cmd/`, `internal/`, `go.mod`, `go.sum`, `Dockerfile`) therefore has to bump, in the same PR: + +1. `appVersion` and `version` in `charts/openfga-operator/Chart.yaml` +2. the `openfga-operator` dependency version and `version` in `charts/openfga/Chart.yaml`, then `helm dependency update charts/openfga` to refresh `Chart.lock` + +`.github/scripts/check-operator-release.sh origin/main` checks the first two files and runs on every PR; `helm dependency build` fails when `Chart.lock` is stale. + +## Local Testing + +Integration test values and instructions are in [`tests/`](tests/). Three scenarios are provided: + +| Scenario | Values File | What It Tests | +|----------|-------------|---------------| +| Happy path | `tests/values-happy-path.yaml` | Full lifecycle: Postgres up, migration succeeds, OpenFGA ready at 3/3 | +| DB outage & recovery | `tests/values-db-outage.yaml` | Postgres starts at 0 replicas; scale it up later to verify self-healing | +| No database | `tests/values-no-db.yaml` | Permanent failure: operator retries without crashing; the app pods stay NotReady (0/3) | + +Quick start: + +```bash +# 1. Build the operator image +cd operator +docker build -t openfga/openfga-operator:dev . + +# 2. Update chart dependencies +cd .. +helm dependency update charts/openfga + +# 3. Run the happy-path test +kubectl create namespace openfga-test +helm install openfga-test charts/openfga -n openfga-test \ + -f operator/tests/values-happy-path.yaml + +# 4. Verify (wait ~30s) +kubectl get all -n openfga-test + +# 5. Clean up +helm uninstall openfga-test -n openfga-test +kubectl delete namespace openfga-test +``` + +See [`tests/README.md`](tests/README.md) for detailed verification steps and all three scenarios. + +## Project Structure + +```text +operator/ +├── cmd/ +│ └── main.go # Entry point, manager setup +├── internal/ +│ └── controller/ +│ ├── migration_controller.go # Reconciliation loop +│ ├── migration_controller_test.go # Unit tests +│ └── helpers.go # Job builder, status ConfigMap helpers +├── Dockerfile # Multi-stage build (distroless runtime) +├── Makefile +├── go.mod +└── go.sum +``` + +## Configuration + +The operator accepts the following flags: + +| Flag | Default | Description | +|------|---------|-------------| +| `--leader-elect` | `false` | Enable leader election so only one replica actively reconciles at a time. Required when running multiple operator replicas for high availability; standby pods wait for the leader's Lease to expire before taking over. Not needed for single-replica deployments. | +| `--watch-namespace` | `""` | Namespace to watch for OpenFGA Deployments. Defaults to the operator pod's own namespace (via `POD_NAMESPACE` env var). The chart binds namespaced RBAC in the configured watch namespace, so the operator may run in a different namespace when needed. | +| `--metrics-bind-address` | `:8080` | Address the Prometheus metrics endpoint binds to; `0` disables it. The chart passes `0` unless `metrics.enabled` is set, which also declares a `metrics` container port. The endpoint is plain HTTP without authentication. | +| `--health-probe-bind-address` | `:8081` | Address the Kubernetes liveness and readiness probe endpoints bind to. Change only if the default port conflicts. | +| `--backoff-limit` | `3` | Number of times a migration Job's pod can fail before the Job is considered failed. The operator then sets a `MigrationFailed` condition on the Deployment and replaces the Job 60 seconds after it failed. | +| `--active-deadline-seconds` | `0` | Maximum wall-clock seconds a migration Job can run before Kubernetes terminates it. `0` means no deadline. A deadline cuts off long migrations, such as index builds or MySQL table rebuilds on large tables, which then start over on the next attempt. | +| `--ttl-seconds-after-finished` | `300` | Seconds Kubernetes keeps a completed Job and its pod before garbage-collecting them. Failed Jobs do not receive a TTL and remain available for the operator's 60-second retry delay. | + +When deployed via the Helm subchart, these are configured through `values.yaml`. See `charts/openfga-operator/values.yaml` for all available options. + +## Annotations + +The operator reads these annotations from the OpenFGA Deployment: + +| Annotation | Description | +|------------|-------------| +| `openfga.dev/migration-enabled` | Must be `"true"` for the operator to manage migrations. Deployments without this annotation are ignored. Set by the Helm chart when `openfga-operator.enabled` and `datastore.applyMigrations` are true and the datastore is Postgres or MySQL. | +| `openfga.dev/container-name` | The OpenFGA container in the pod spec. Defaults to `openfga`. | +| `openfga.dev/migration-trigger` | Any string that is part of the migration identity alongside the image tag; a change runs the migration again. The chart derives it from the datastore settings and `migration.trigger`. | +| `openfga.dev/migration-service-account` | The ServiceAccount to use for migration Jobs. Defaults to the Deployment's SA. | +| `openfga.dev/migration-init-containers` | JSON array of additional init containers for the migration Job. Generated from `migrate.extraInitContainers`. | +| `openfga.dev/migration-sidecars` | JSON array of additional native sidecars for the migration Job. Generated from `migrate.sidecars`. | +| `openfga.dev/migration-volumes` | JSON array of additional volumes for the migration Job. Generated from `migrate.extraVolumes`. | +| `openfga.dev/migration-volume-mounts` | JSON array of additional mounts for the migration container. Generated from `migrate.extraVolumeMounts`. | +| `openfga.dev/migration-resources` | JSON resource requirements for the migration container. Generated from `datastore.migrations.resources`. | +| `openfga.dev/migration-timeout` | `OPENFGA_TIMEOUT` for the migration container. Generated from `migrate.timeout`. | +| `openfga.dev/migration-annotations` | JSON map of non-Helm annotations for the migration Job and pod. Generated from `migrate.annotations`; `helm.sh/*` hook annotations are excluded. | +| `openfga.dev/migration-labels` | JSON map of additional labels for the migration Job and pod. Generated from `migrate.labels`; operator identity labels take precedence. | + +## Security + +The operator is namespace-scoped. It watches one namespace, runs with a Role rather than a ClusterRole, and never reads Secrets itself: the migration Job gets the OpenFGA container's environment, including `secretKeyRef` entries, and runs as the service account named in `openfga.dev/migration-service-account`, or the OpenFGA pod's service account when that annotation is absent. + +The Role the chart creates in the watch namespace: + +| Resource | Verbs | Used for | +|----------|-------|----------| +| `apps/deployments` | get, list, watch | Find opted-in OpenFGA Deployments | +| `apps/deployments/status` | patch | Set the `MigrationFailed` condition | +| `batch/jobs` | get, list, watch, create, delete, patch | Run, replace and expire migration Jobs | +| `configmaps` | get, list, watch, create, update | Record the migrated version | +| `coordination.k8s.io/leases` | get, list, watch, create, update | Leader election | +| `events` | create, patch | `MigrationStarted`, `MigrationSucceeded`, `MigrationFailed`, `MigrationJobConflict` | + +Ports: `8081` serves `/healthz` and `/readyz` for the kubelet. `8080` serves Prometheus metrics only when `metrics.enabled` is set; it has no authentication, so restrict it with a NetworkPolicy if the namespace is shared. + +Images pushed by `.github/workflows/operator.yml` carry an SBOM and build provenance and are signed with cosign keyless. To verify: + +```bash +cosign verify ghcr.io/openfga/openfga-operator:<tag> \ + --certificate-oidc-issuer https://token.actions.githubusercontent.com \ + --certificate-identity-regexp '^https://github.com/openfga/helm-charts/\.github/workflows/operator\.yml@refs/heads/' +``` + +Report vulnerabilities through the [security policy](https://github.com/openfga/helm-charts/security/policy), not in a public issue. + +## Limitations + +- **Secret contents are not observable:** The trigger covers the datastore settings the chart renders (engine, database host and name, Secret names and keys) but not the contents of a Secret that changes under the same name; set `migration.trigger` to a new value in that case. Credentials in the URI are never part of the trigger. +- **Mutable image contents are not observable:** Reusing a tag such as `latest` does not change the Deployment's image reference. Use immutable tags or digests, or change `migration.trigger` when deliberately replacing the contents of a mutable tag. +- **Helm hook metadata:** Operator-managed Jobs ignore `helm.sh/*` entries in `migrate.annotations`. Other migration annotations and labels are forwarded, but cannot override the operator's identity labels. +- **Injected sidecars:** Containers injected by a webhook are not part of the Deployment's pod spec and cannot be converted to native sidecars. Disable injection for the migration pod with `migrate.annotations` if the injected container does not exit. +- **Job pod labels:** The migration pod is labelled `app.kubernetes.io/part-of: openfga` and `app.kubernetes.io/component: migration`, not with the OpenFGA Deployment's `app.kubernetes.io/name`/`instance` labels (which would make it a Service endpoint). A NetworkPolicy that allows database egress only for the OpenFGA pods' labels needs a rule for the migration pod too. +- **Same-name resources:** The operator only replaces a `{name}-migrate` Job it created itself or the chart's legacy hook Job, and only trusts a `{name}-migration-status` ConfigMap owned by the Deployment. Anything else with those names blocks the migration with a `MigrationJobConflict` event until it is removed. +- **One namespace per operator:** The operator reconciles every opted-in OpenFGA Deployment in its watch namespace. Operators installed by several releases in one namespace share a leader election lease, so only one of them is active at a time. diff --git a/operator/cmd/main.go b/operator/cmd/main.go new file mode 100644 index 00000000..40f96fa1 --- /dev/null +++ b/operator/cmd/main.go @@ -0,0 +1,125 @@ +package main + +import ( + "flag" + "fmt" + "math" + "os" + + "k8s.io/apimachinery/pkg/runtime" + clientgoscheme "k8s.io/client-go/kubernetes/scheme" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/cache" + "sigs.k8s.io/controller-runtime/pkg/healthz" + "sigs.k8s.io/controller-runtime/pkg/log/zap" + metricsserver "sigs.k8s.io/controller-runtime/pkg/metrics/server" + + "github.com/openfga/helm-charts/operator/internal/controller" +) + +var scheme = runtime.NewScheme() + +func init() { + _ = clientgoscheme.AddToScheme(scheme) +} + +func main() { + var ( + leaderElect bool + watchNamespace string + metricsAddr string + healthProbeAddr string + backoffLimit int + activeDeadline int + ttlAfterFinished int + ) + + flag.BoolVar(&leaderElect, "leader-elect", false, "Enable leader election for the controller manager.") + flag.StringVar(&watchNamespace, "watch-namespace", "", "Namespace to watch. Defaults to the operator pod namespace.") + flag.StringVar(&metricsAddr, "metrics-bind-address", ":8080", "The address the metric endpoint binds to.") + flag.StringVar(&healthProbeAddr, "health-probe-bind-address", ":8081", "The address the health probe endpoint binds to.") + flag.IntVar(&backoffLimit, "backoff-limit", int(controller.DefaultBackoffLimit), "BackoffLimit for migration Jobs.") + flag.IntVar(&activeDeadline, "active-deadline-seconds", int(controller.DefaultActiveDeadlineSeconds), "ActiveDeadlineSeconds for migration Jobs; 0 means no deadline.") + flag.IntVar(&ttlAfterFinished, "ttl-seconds-after-finished", int(controller.DefaultTTLSecondsAfterFinished), "TTLSecondsAfterFinished for migration Jobs.") + + opts := zap.Options{Development: false} + opts.BindFlags(flag.CommandLine) + flag.Parse() + + // Validate flag values. + for _, v := range []struct { + name string + value int + max int + }{ + {"backoff-limit", backoffLimit, math.MaxInt32}, + {"active-deadline-seconds", activeDeadline, math.MaxInt32}, + {"ttl-seconds-after-finished", ttlAfterFinished, math.MaxInt32}, + } { + if v.value < 0 || v.value > v.max { + fmt.Fprintf(os.Stderr, "invalid value for --%s: must be between 0 and %d\n", v.name, v.max) + os.Exit(1) + } + } + + ctrl.SetLogger(zap.New(zap.UseFlagOptions(&opts))) + logger := ctrl.Log.WithName("setup") + + // Fall back to the pod's namespace when no explicit scope is set. + if watchNamespace == "" { + if podNS, ok := os.LookupEnv("POD_NAMESPACE"); ok && podNS != "" { + watchNamespace = podNS + logger.Info("defaulting watch scope to pod namespace", "namespace", podNS) + } + } + + // Configure cache namespace restrictions. + var cacheOpts cache.Options + if watchNamespace != "" { + cacheOpts.DefaultNamespaces = map[string]cache.Config{ + watchNamespace: {}, + } + } + + mgr, err := ctrl.NewManager(ctrl.GetConfigOrDie(), ctrl.Options{ + Scheme: scheme, + Metrics: metricsserver.Options{BindAddress: metricsAddr}, + HealthProbeBindAddress: healthProbeAddr, + LeaderElection: leaderElect, + LeaderElectionID: "openfga-operator-leader", + LeaderElectionNamespace: watchNamespace, + Cache: cacheOpts, + }) + if err != nil { + logger.Error(err, "unable to create manager") + os.Exit(1) + } + + reconciler := &controller.MigrationReconciler{ + Client: mgr.GetClient(), + Recorder: mgr.GetEventRecorderFor("openfga-operator"), + BackoffLimit: int32(backoffLimit), + ActiveDeadlineSeconds: int64(activeDeadline), + TTLSecondsAfterFinished: int32(ttlAfterFinished), + } + + if err := reconciler.SetupWithManager(mgr); err != nil { + logger.Error(err, "unable to create controller", "controller", "MigrationReconciler") + os.Exit(1) + } + + if err := mgr.AddHealthzCheck("healthz", healthz.Ping); err != nil { + logger.Error(err, "unable to set up health check") + os.Exit(1) + } + if err := mgr.AddReadyzCheck("readyz", healthz.Ping); err != nil { + logger.Error(err, "unable to set up readiness check") + os.Exit(1) + } + + logger.Info("starting manager") + if err := mgr.Start(ctrl.SetupSignalHandler()); err != nil { + logger.Error(err, "problem running manager") + os.Exit(1) + } +} diff --git a/operator/go.mod b/operator/go.mod new file mode 100644 index 00000000..7265283c --- /dev/null +++ b/operator/go.mod @@ -0,0 +1,66 @@ +module github.com/openfga/helm-charts/operator + +go 1.26.8 + +require ( + k8s.io/api v0.35.3 + k8s.io/apimachinery v0.35.3 + k8s.io/client-go v0.35.3 + k8s.io/utils v0.0.0-20260319190234-28399d86e0b5 + sigs.k8s.io/controller-runtime v0.23.3 +) + +require ( + github.com/beorn7/perks v1.0.1 // indirect + github.com/cespare/xxhash/v2 v2.3.0 // indirect + github.com/davecgh/go-spew v1.1.1 // indirect + github.com/emicklei/go-restful/v3 v3.12.2 // indirect + github.com/evanphx/json-patch/v5 v5.9.11 // indirect + github.com/fsnotify/fsnotify v1.9.0 // indirect + github.com/fxamacker/cbor/v2 v2.9.0 // indirect + github.com/go-logr/logr v1.4.3 // indirect + github.com/go-logr/zapr v1.3.0 // indirect + github.com/go-openapi/jsonpointer v0.21.0 // indirect + github.com/go-openapi/jsonreference v0.20.2 // indirect + github.com/go-openapi/swag v0.23.0 // indirect + github.com/google/btree v1.1.3 // indirect + github.com/google/gnostic-models v0.7.0 // indirect + github.com/google/go-cmp v0.7.0 // indirect + github.com/google/uuid v1.6.0 // indirect + github.com/josharian/intern v1.0.0 // indirect + github.com/json-iterator/go v1.1.12 // indirect + github.com/mailru/easyjson v0.7.7 // indirect + github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect + github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee // indirect + github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect + github.com/pmezard/go-difflib v1.0.0 // indirect + github.com/prometheus/client_golang v1.23.2 // indirect + github.com/prometheus/client_model v0.6.2 // indirect + github.com/prometheus/common v0.66.1 // indirect + github.com/prometheus/procfs v0.16.1 // indirect + github.com/spf13/pflag v1.0.9 // indirect + github.com/x448/float16 v0.8.4 // indirect + go.uber.org/multierr v1.11.0 // indirect + go.uber.org/zap v1.27.0 // indirect + go.yaml.in/yaml/v2 v2.4.3 // indirect + go.yaml.in/yaml/v3 v3.0.4 // indirect + golang.org/x/net v0.59.0 // indirect + golang.org/x/oauth2 v0.30.0 // indirect + golang.org/x/sync v0.23.0 // indirect + golang.org/x/sys v0.48.0 // indirect + golang.org/x/term v0.46.0 // indirect + golang.org/x/text v0.42.0 // indirect + golang.org/x/time v0.9.0 // indirect + gomodules.xyz/jsonpatch/v2 v2.4.0 // indirect + google.golang.org/protobuf v1.36.8 // indirect + gopkg.in/evanphx/json-patch.v4 v4.13.0 // indirect + gopkg.in/inf.v0 v0.9.1 // indirect + gopkg.in/yaml.v3 v3.0.1 // indirect + k8s.io/apiextensions-apiserver v0.35.0 // indirect + k8s.io/klog/v2 v2.130.1 // indirect + k8s.io/kube-openapi v0.0.0-20250910181357-589584f1c912 // indirect + sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 // indirect + sigs.k8s.io/randfill v1.0.0 // indirect + sigs.k8s.io/structured-merge-diff/v6 v6.3.2-0.20260122202528-d9cc6641c482 // indirect + sigs.k8s.io/yaml v1.6.0 // indirect +) diff --git a/operator/go.sum b/operator/go.sum new file mode 100644 index 00000000..b321fc24 --- /dev/null +++ b/operator/go.sum @@ -0,0 +1,171 @@ +github.com/Masterminds/semver/v3 v3.4.0 h1:Zog+i5UMtVoCU8oKka5P7i9q9HgrJeGzI9SA1Xbatp0= +github.com/Masterminds/semver/v3 v3.4.0/go.mod h1:4V+yj/TJE1HU9XfppCwVMZq3I84lprf4nC11bSS5beM= +github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM= +github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw= +github.com/cespare/xxhash/v2 v2.3.0 h1:UL815xU9SqsFlibzuggzjXhog7bL6oX9BbNZnL2UFvs= +github.com/cespare/xxhash/v2 v2.3.0/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs= +github.com/creack/pty v1.1.9/go.mod h1:oKZEueFk5CKHvIhNR5MUki03XCEU+Q6VDXinZuGJ33E= +github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= +github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c= +github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= +github.com/emicklei/go-restful/v3 v3.12.2 h1:DhwDP0vY3k8ZzE0RunuJy8GhNpPL6zqLkDf9B/a0/xU= +github.com/emicklei/go-restful/v3 v3.12.2/go.mod h1:6n3XBCmQQb25CM2LCACGz8ukIrRry+4bhvbpWn3mrbc= +github.com/evanphx/json-patch v0.5.2 h1:xVCHIVMUu1wtM/VkR9jVZ45N3FhZfYMMYGorLCR8P3k= +github.com/evanphx/json-patch v0.5.2/go.mod h1:ZWS5hhDbVDyob71nXKNL0+PWn6ToqBHMikGIFbs31qQ= +github.com/evanphx/json-patch/v5 v5.9.11 h1:/8HVnzMq13/3x9TPvjG08wUGqBTmZBsCWzjTM0wiaDU= +github.com/evanphx/json-patch/v5 v5.9.11/go.mod h1:3j+LviiESTElxA4p3EMKAB9HXj3/XEtnUf6OZxqIQTM= +github.com/fsnotify/fsnotify v1.9.0 h1:2Ml+OJNzbYCTzsxtv8vKSFD9PbJjmhYF14k/jKC7S9k= +github.com/fsnotify/fsnotify v1.9.0/go.mod h1:8jBTzvmWwFyi3Pb8djgCCO5IBqzKJ/Jwo8TRcHyHii0= +github.com/fxamacker/cbor/v2 v2.9.0 h1:NpKPmjDBgUfBms6tr6JZkTHtfFGcMKsw3eGcmD/sapM= +github.com/fxamacker/cbor/v2 v2.9.0/go.mod h1:vM4b+DJCtHn+zz7h3FFp/hDAI9WNWCsZj23V5ytsSxQ= +github.com/go-logr/logr v1.4.3 h1:CjnDlHq8ikf6E492q6eKboGOC0T8CDaOvkHCIg8idEI= +github.com/go-logr/logr v1.4.3/go.mod h1:9T104GzyrTigFIr8wt5mBrctHMim0Nb2HLGrmQ40KvY= +github.com/go-logr/zapr v1.3.0 h1:XGdV8XW8zdwFiwOA2Dryh1gj2KRQyOOoNmBy4EplIcQ= +github.com/go-logr/zapr v1.3.0/go.mod h1:YKepepNBd1u/oyhd/yQmtjVXmm9uML4IXUgMOwR8/Gg= +github.com/go-openapi/jsonpointer v0.19.6/go.mod h1:osyAmYz/mB/C3I+WsTTSgw1ONzaLJoLCyoi6/zppojs= +github.com/go-openapi/jsonpointer v0.21.0 h1:YgdVicSA9vH5RiHs9TZW5oyafXZFc6+2Vc1rr/O9oNQ= +github.com/go-openapi/jsonpointer v0.21.0/go.mod h1:IUyH9l/+uyhIYQ/PXVA41Rexl+kOkAPDdXEYns6fzUY= +github.com/go-openapi/jsonreference v0.20.2 h1:3sVjiK66+uXK/6oQ8xgcRKcFgQ5KXa2KvnJRumpMGbE= +github.com/go-openapi/jsonreference v0.20.2/go.mod h1:Bl1zwGIM8/wsvqjsOQLJ/SH+En5Ap4rVB5KVcIDZG2k= +github.com/go-openapi/swag v0.22.3/go.mod h1:UzaqsxGiab7freDnrUUra0MwWfN/q7tE4j+VcZ0yl14= +github.com/go-openapi/swag v0.23.0 h1:vsEVJDUo2hPJ2tu0/Xc+4noaxyEffXNIs3cOULZ+GrE= +github.com/go-openapi/swag v0.23.0/go.mod h1:esZ8ITTYEsH1V2trKHjAN8Ai7xHb8RV+YSZ577vPjgQ= +github.com/go-task/slim-sprig/v3 v3.0.0 h1:sUs3vkvUymDpBKi3qH1YSqBQk9+9D/8M2mN1vB6EwHI= +github.com/go-task/slim-sprig/v3 v3.0.0/go.mod h1:W848ghGpv3Qj3dhTPRyJypKRiqCdHZiAzKg9hl15HA8= +github.com/google/btree v1.1.3 h1:CVpQJjYgC4VbzxeGVHfvZrv1ctoYCAI8vbl07Fcxlyg= +github.com/google/btree v1.1.3/go.mod h1:qOPhT0dTNdNzV6Z/lhRX0YXUafgPLFUh+gZMl761Gm4= +github.com/google/gnostic-models v0.7.0 h1:qwTtogB15McXDaNqTZdzPJRHvaVJlAl+HVQnLmJEJxo= +github.com/google/gnostic-models v0.7.0/go.mod h1:whL5G0m6dmc5cPxKc5bdKdEN3UjI7OUGxBlw57miDrQ= +github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= +github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= +github.com/google/gofuzz v1.0.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg= +github.com/google/gofuzz v1.2.0 h1:xRy4A+RhZaiKjJ1bPfwQ8sedCA+YS2YcCHW6ec7JMi0= +github.com/google/gofuzz v1.2.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg= +github.com/google/pprof v0.0.0-20250403155104-27863c87afa6 h1:BHT72Gu3keYf3ZEu2J0b1vyeLSOYI8bm5wbJM/8yDe8= +github.com/google/pprof v0.0.0-20250403155104-27863c87afa6/go.mod h1:boTsfXsheKC2y+lKOCMpSfarhxDeIzfZG1jqGcPl3cA= +github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= +github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= +github.com/josharian/intern v1.0.0 h1:vlS4z54oSdjm0bgjRigI+G1HpF+tI+9rE5LLzOg8HmY= +github.com/josharian/intern v1.0.0/go.mod h1:5DoeVV0s6jJacbCEi61lwdGj/aVlrQvzHFFd8Hwg//Y= +github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnrnM= +github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo= +github.com/klauspost/compress v1.18.0 h1:c/Cqfb0r+Yi+JtIEq73FWXVkRonBlf0CRNYc8Zttxdo= +github.com/klauspost/compress v1.18.0/go.mod h1:2Pp+KzxcywXVXMr50+X0Q/Lsb43OQHYWRCY2AiWywWQ= +github.com/kr/pretty v0.2.1/go.mod h1:ipq/a2n7PKx3OHsz4KJII5eveXtPO4qwEXGdVfWzfnI= +github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE= +github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk= +github.com/kr/pty v1.1.1/go.mod h1:pFQYn66WHrOpPYNljwOMqo10TkYh1fy3cYio2l3bCsQ= +github.com/kr/text v0.1.0/go.mod h1:4Jbv+DJW3UT/LiOwJeYQe1efqtUx/iVham/4vfdArNI= +github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= +github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE= +github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc= +github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw= +github.com/mailru/easyjson v0.7.7 h1:UGYAvKxe3sBsEDzO8ZeWOSlIQfWFlxbzLZe7hwFURr0= +github.com/mailru/easyjson v0.7.7/go.mod h1:xzfreul335JAWq5oZzymOObrkdz5UnU4kGfJJLY9Nlc= +github.com/modern-go/concurrent v0.0.0-20180228061459-e0a39a4cb421/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q= +github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd h1:TRLaZ9cD/w8PVh93nsPXa1VrQ6jlwL5oN8l14QlcNfg= +github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q= +github.com/modern-go/reflect2 v1.0.2/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk= +github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee h1:W5t00kpgFdJifH4BDsTlE89Zl93FEloxaWZfGcifgq8= +github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk= +github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA= +github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ= +github.com/onsi/ginkgo/v2 v2.27.2 h1:LzwLj0b89qtIy6SSASkzlNvX6WktqurSHwkk2ipF/Ns= +github.com/onsi/ginkgo/v2 v2.27.2/go.mod h1:ArE1D/XhNXBXCBkKOLkbsb2c81dQHCRcF5zwn/ykDRo= +github.com/onsi/gomega v1.38.2 h1:eZCjf2xjZAqe+LeWvKb5weQ+NcPwX84kqJ0cZNxok2A= +github.com/onsi/gomega v1.38.2/go.mod h1:W2MJcYxRGV63b418Ai34Ud0hEdTVXq9NW9+Sx6uXf3k= +github.com/pkg/errors v0.9.1 h1:FEBLx1zS214owpjy7qsBeixbURkuhQAwrK5UwLGTwt4= +github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINEl0= +github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM= +github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= +github.com/prometheus/client_golang v1.23.2 h1:Je96obch5RDVy3FDMndoUsjAhG5Edi49h0RJWRi/o0o= +github.com/prometheus/client_golang v1.23.2/go.mod h1:Tb1a6LWHB3/SPIzCoaDXI4I8UHKeFTEQ1YCr+0Gyqmg= +github.com/prometheus/client_model v0.6.2 h1:oBsgwpGs7iVziMvrGhE53c/GrLUsZdHnqNwqPLxwZyk= +github.com/prometheus/client_model v0.6.2/go.mod h1:y3m2F6Gdpfy6Ut/GBsUqTWZqCUvMVzSfMLjcu6wAwpE= +github.com/prometheus/common v0.66.1 h1:h5E0h5/Y8niHc5DlaLlWLArTQI7tMrsfQjHV+d9ZoGs= +github.com/prometheus/common v0.66.1/go.mod h1:gcaUsgf3KfRSwHY4dIMXLPV0K/Wg1oZ8+SbZk/HH/dA= +github.com/prometheus/procfs v0.16.1 h1:hZ15bTNuirocR6u0JZ6BAHHmwS1p8B4P6MRqxtzMyRg= +github.com/prometheus/procfs v0.16.1/go.mod h1:teAbpZRB1iIAJYREa1LsoWUXykVXA1KlTmWl8x/U+Is= +github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ= +github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc= +github.com/spf13/pflag v1.0.9 h1:9exaQaMOCwffKiiiYk6/BndUBv+iRViNW+4lEMi0PvY= +github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= +github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME= +github.com/stretchr/objx v0.4.0/go.mod h1:YvHI0jy2hoMjB+UWwv71VJQ9isScKT/TqJzVSSt89Yw= +github.com/stretchr/objx v0.5.0/go.mod h1:Yh+to48EsGEfYuaHDzXPcE3xhTkx73EhmCGUpEOglKo= +github.com/stretchr/objx v0.5.2 h1:xuMeJ0Sdp5ZMRXx/aWO6RZxdr3beISkG5/G/aIRr3pY= +github.com/stretchr/objx v0.5.2/go.mod h1:FRsXN1f5AsAjCGJKqEizvkpNtU+EGNCLh3NxZ/8L+MA= +github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI= +github.com/stretchr/testify v1.7.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= +github.com/stretchr/testify v1.8.0/go.mod h1:yNjHg4UonilssWZ8iaSj1OCr/vHnekPRkoO+kdMU+MU= +github.com/stretchr/testify v1.8.1/go.mod h1:w2LPCIKwWwSfY2zedu0+kehJoqGctiVI29o6fzry7u4= +github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U= +github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U= +github.com/x448/float16 v0.8.4 h1:qLwI1I70+NjRFUR3zs1JPUCgaCXSh3SW62uAKT1mSBM= +github.com/x448/float16 v0.8.4/go.mod h1:14CWIYCyZA/cWjXOioeEpHeN/83MdbZDRQHoFcYsOfg= +go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto= +go.uber.org/goleak v1.3.0/go.mod h1:CoHD4mav9JJNrW/WLlf7HGZPjdw8EucARQHekz1X6bE= +go.uber.org/multierr v1.11.0 h1:blXXJkSxSSfBVBlC76pxqeO+LN3aDfLQo+309xJstO0= +go.uber.org/multierr v1.11.0/go.mod h1:20+QtiLqy0Nd6FdQB9TLXag12DsQkrbs3htMFfDN80Y= +go.uber.org/zap v1.27.0 h1:aJMhYGrd5QSmlpLMr2MftRKl7t8J8PTZPA732ud/XR8= +go.uber.org/zap v1.27.0/go.mod h1:GB2qFLM7cTU87MWRP2mPIjqfIDnGu+VIO4V/SdhGo2E= +go.yaml.in/yaml/v2 v2.4.3 h1:6gvOSjQoTB3vt1l+CU+tSyi/HOjfOjRLJ4YwYZGwRO0= +go.yaml.in/yaml/v2 v2.4.3/go.mod h1:zSxWcmIDjOzPXpjlTTbAsKokqkDNAVtZO0WOMiT90s8= +go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc= +go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg= +golang.org/x/mod v0.41.0 h1:qJmnOUb4YB+FsEuM3HcWucdZASCPGhsX6uljO6pog0c= +golang.org/x/mod v0.41.0/go.mod h1:Ek9pY8RKWXwsWvd3rQiHYtMqkjSUV+s1Rj7j4H5Ur6o= +golang.org/x/net v0.59.0 h1:5zfYln+w5XCxwrnMMJPufRgNoXEaGxl0wo5GqPXyues= +golang.org/x/net v0.59.0/go.mod h1:2DA/G1UfVbCpQPeWTmMPGY7Cs2PkBkwu743bVX5PIVg= +golang.org/x/oauth2 v0.30.0 h1:dnDm7JmhM45NNpd8FDDeLhK6FwqbOf4MLCM9zb1BOHI= +golang.org/x/oauth2 v0.30.0/go.mod h1:B++QgG3ZKulg6sRPGD/mqlHQs5rB3Ml9erfeDY7xKlU= +golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk= +golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0= +golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo= +golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og= +golang.org/x/term v0.46.0 h1:3+OXuTbaKDgwk8jTi3aSLHRlmWqHEUDUtxnbFigO4YE= +golang.org/x/term v0.46.0/go.mod h1:+K02xbkittuwc0Am4abfA3Fc+XRGXkvBXNO88NCXPoc= +golang.org/x/text v0.42.0 h1:JbOZXgfeCPU9gacVtYliJqOhD+zhrEqK4LfdpmlUZqI= +golang.org/x/text v0.42.0/go.mod h1:ojzP1Z+2QtioaF8DTtO8K5q7JWVVYwZKenzujK0Zd0E= +golang.org/x/time v0.9.0 h1:EsRrnYcQiGH+5FfbgvV4AP7qEZstoyrHB0DzarOQ4ZY= +golang.org/x/time v0.9.0/go.mod h1:3BpzKBy/shNhVucY/MWOyx10tF3SFh9QdLuxbVysPQM= +golang.org/x/tools v0.49.0 h1:3NI7VXzL9+1WZD52Dx2ttoPwD5DWrFGpl9mFZDlmisI= +golang.org/x/tools v0.49.0/go.mod h1:SJNXV9DBKT0UbdttsQjbfJlAE/q+y36++zo3uL3N0Oo= +gomodules.xyz/jsonpatch/v2 v2.4.0 h1:Ci3iUJyx9UeRx7CeFN8ARgGbkESwJK+KB9lLcWxY/Zw= +gomodules.xyz/jsonpatch/v2 v2.4.0/go.mod h1:AH3dM2RI6uoBZxn3LVrfvJ3E0/9dG4cSrbuBJT4moAY= +google.golang.org/protobuf v1.36.8 h1:xHScyCOEuuwZEc6UtSOvPbAT4zRh0xcNRYekJwfqyMc= +google.golang.org/protobuf v1.36.8/go.mod h1:fuxRtAxBytpl4zzqUh6/eyUujkJdNiuEkXntxiD/uRU= +gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= +gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk= +gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q= +gopkg.in/evanphx/json-patch.v4 v4.13.0 h1:czT3CmqEaQ1aanPc5SdlgQrrEIb8w/wwCvWWnfEbYzo= +gopkg.in/evanphx/json-patch.v4 v4.13.0/go.mod h1:p8EYWUEYMpynmqDbY58zCKCFZw8pRWMG4EsWvDvM72M= +gopkg.in/inf.v0 v0.9.1 h1:73M5CoZyi3ZLMOyDlQh031Cx6N9NDJ2Vvfl76EDAgDc= +gopkg.in/inf.v0 v0.9.1/go.mod h1:cWUDdTG/fYaXco+Dcufb5Vnc6Gp2YChqWtbxRZE0mXw= +gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= +gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +k8s.io/api v0.35.3 h1:pA2fiBc6+N9PDf7SAiluKGEBuScsTzd2uYBkA5RzNWQ= +k8s.io/api v0.35.3/go.mod h1:9Y9tkBcFwKNq2sxwZTQh1Njh9qHl81D0As56tu42GA4= +k8s.io/apiextensions-apiserver v0.35.0 h1:3xHk2rTOdWXXJM+RDQZJvdx0yEOgC0FgQ1PlJatA5T4= +k8s.io/apiextensions-apiserver v0.35.0/go.mod h1:E1Ahk9SADaLQ4qtzYFkwUqusXTcaV2uw3l14aqpL2LU= +k8s.io/apimachinery v0.35.3 h1:MeaUwQCV3tjKP4bcwWGgZ/cp/vpsRnQzqO6J6tJyoF8= +k8s.io/apimachinery v0.35.3/go.mod h1:jQCgFZFR1F4Ik7hvr2g84RTJSZegBc8yHgFWKn//hns= +k8s.io/client-go v0.35.3 h1:s1lZbpN4uI6IxeTM2cpdtrwHcSOBML1ODNTCCfsP1pg= +k8s.io/client-go v0.35.3/go.mod h1:RzoXkc0mzpWIDvBrRnD+VlfXP+lRzqQjCmKtiwZ8Q9c= +k8s.io/klog/v2 v2.130.1 h1:n9Xl7H1Xvksem4KFG4PYbdQCQxqc/tTUyrgXaOhHSzk= +k8s.io/klog/v2 v2.130.1/go.mod h1:3Jpz1GvMt720eyJH1ckRHK1EDfpxISzJ7I9OYgaDtPE= +k8s.io/kube-openapi v0.0.0-20250910181357-589584f1c912 h1:Y3gxNAuB0OBLImH611+UDZcmKS3g6CthxToOb37KgwE= +k8s.io/kube-openapi v0.0.0-20250910181357-589584f1c912/go.mod h1:kdmbQkyfwUagLfXIad1y2TdrjPFWp2Q89B3qkRwf/pQ= +k8s.io/utils v0.0.0-20260319190234-28399d86e0b5 h1:kBawHLSnx/mYHmRnNUf9d4CpjREbeZuxoSGOX/J+aYM= +k8s.io/utils v0.0.0-20260319190234-28399d86e0b5/go.mod h1:xDxuJ0whA3d0I4mf/C4ppKHxXynQ+fxnkmQH0vTHnuk= +sigs.k8s.io/controller-runtime v0.23.3 h1:VjB/vhoPoA9l1kEKZHBMnQF33tdCLQKJtydy4iqwZ80= +sigs.k8s.io/controller-runtime v0.23.3/go.mod h1:B6COOxKptp+YaUT5q4l6LqUJTRpizbgf9KSRNdQGns0= +sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 h1:IpInykpT6ceI+QxKBbEflcR5EXP7sU1kvOlxwZh5txg= +sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730/go.mod h1:mdzfpAEoE6DHQEN0uh9ZbOCuHbLK5wOm7dK4ctXE9Tg= +sigs.k8s.io/randfill v1.0.0 h1:JfjMILfT8A6RbawdsK2JXGBR5AQVfd+9TbzrlneTyrU= +sigs.k8s.io/randfill v1.0.0/go.mod h1:XeLlZ/jmk4i1HRopwe7/aU3H5n1zNUcX6TM94b3QxOY= +sigs.k8s.io/structured-merge-diff/v6 v6.3.2-0.20260122202528-d9cc6641c482 h1:2WOzJpHUBVrrkDjU4KBT8n5LDcj824eX0I5UKcgeRUs= +sigs.k8s.io/structured-merge-diff/v6 v6.3.2-0.20260122202528-d9cc6641c482/go.mod h1:M3W8sfWvn2HhQDIbGWj3S099YozAsymCo/wrT5ohRUE= +sigs.k8s.io/yaml v1.6.0 h1:G8fkbMSAFqgEFgh4b1wmtzDnioxFCUgTZhlbj5P9QYs= +sigs.k8s.io/yaml v1.6.0/go.mod h1:796bPqUfzR/0jLAl6XjHl3Ck7MiyVv8dbTdyT3/pMf4= diff --git a/operator/internal/controller/helpers.go b/operator/internal/controller/helpers.go new file mode 100644 index 00000000..410e6166 --- /dev/null +++ b/operator/internal/controller/helpers.go @@ -0,0 +1,429 @@ +package controller + +import ( + "context" + "crypto/sha256" + "encoding/json" + "fmt" + "io" + "strings" + "time" + + appsv1 "k8s.io/api/apps/v1" + batchv1 "k8s.io/api/batch/v1" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/controller/controllerutil" +) + +const ( + // Labels used to discover OpenFGA Deployments. + LabelPartOf = "app.kubernetes.io/part-of" + LabelComponent = "app.kubernetes.io/component" + + LabelPartOfValue = "openfga" + LabelComponentValue = "authorization-controller" + + // Labels set on operator-managed resources (migration Jobs, status ConfigMaps). + LabelManagedBy = "app.kubernetes.io/managed-by" + LabelManagedByValue = "openfga-operator" + + // Annotations read from the Deployment (set by the Helm chart). + AnnotationMigrationEnabled = "openfga.dev/migration-enabled" + AnnotationContainerName = "openfga.dev/container-name" + AnnotationMigrationServiceAccount = "openfga.dev/migration-service-account" + AnnotationMigrationTrigger = "openfga.dev/migration-trigger" + AnnotationMigrationInitContainers = "openfga.dev/migration-init-containers" + AnnotationMigrationSidecars = "openfga.dev/migration-sidecars" + AnnotationMigrationVolumes = "openfga.dev/migration-volumes" + AnnotationMigrationVolumeMounts = "openfga.dev/migration-volume-mounts" + AnnotationMigrationResources = "openfga.dev/migration-resources" + AnnotationMigrationTimeout = "openfga.dev/migration-timeout" + AnnotationMigrationAnnotations = "openfga.dev/migration-annotations" + AnnotationMigrationLabels = "openfga.dev/migration-labels" + + // Annotations set on migration Jobs. + AnnotationDesiredVersion = "openfga.dev/desired-version" + AnnotationPodTemplateHash = "openfga.dev/pod-template-hash" + + // Defaults for migration Job configuration. An ActiveDeadlineSeconds of 0 + // leaves the Job without a deadline. + DefaultBackoffLimit int32 = 3 + DefaultActiveDeadlineSeconds int64 = 0 + DefaultTTLSecondsAfterFinished int32 = 300 +) + +// migrationIdentity is what the operator compares to decide whether a +// migration has to run: the OpenFGA image version plus the trigger the chart +// derives from the datastore configuration. +type migrationIdentity struct { + Version string + Trigger string +} + +func desiredIdentity(deployment *appsv1.Deployment, container *corev1.Container) migrationIdentity { + return migrationIdentity{ + Version: extractImageTag(container.Image), + Trigger: deployment.Annotations[AnnotationMigrationTrigger], + } +} + +func jobIdentity(job *batchv1.Job) migrationIdentity { + return migrationIdentity{ + Version: job.Annotations[AnnotationDesiredVersion], + Trigger: job.Annotations[AnnotationMigrationTrigger], + } +} + +// recordedIdentity returns the identity stored in the status ConfigMap, or the +// zero value when the ConfigMap is missing or not owned by this Deployment. +func recordedIdentity(cm *corev1.ConfigMap, deployment *appsv1.Deployment) migrationIdentity { + if !metav1.IsControlledBy(cm, deployment) { + return migrationIdentity{} + } + return migrationIdentity{Version: cm.Data["version"], Trigger: cm.Data["trigger"]} +} + +// extractImageTag returns the tag portion of a container image reference. +// For "openfga/openfga:v1.14.0" it returns "v1.14.0". +// For "openfga/openfga@sha256:abc..." it returns the digest. +// If there is no tag or digest, it returns "latest". +func extractImageTag(image string) string { + if idx := strings.LastIndex(image, "@"); idx != -1 { + return image[idx+1:] + } + + // Only look for ":" after the last "/" so a registry port is not mistaken for a tag. + nameAndTag := image[strings.LastIndex(image, "/")+1:] + if idx := strings.LastIndex(nameAndTag, ":"); idx != -1 { + return nameAndTag[idx+1:] + } + return "latest" +} + +// migrationConfigMapName returns the name of the ConfigMap used to track migration state. +func migrationConfigMapName(deploymentName string) string { + return deploymentName + "-migration-status" +} + +// migrationJobName returns the name of the migration Job. +func migrationJobName(deploymentName string) string { + return deploymentName + "-migrate" +} + +// findOpenFGAContainer returns the container named by the openfga.dev/container-name +// annotation, or the container named "openfga" when the annotation is absent. +func findOpenFGAContainer(deployment *appsv1.Deployment) (*corev1.Container, error) { + targetName := deployment.Annotations[AnnotationContainerName] + if targetName == "" { + targetName = "openfga" + } + containers := deployment.Spec.Template.Spec.Containers + for i := range containers { + if containers[i].Name == targetName { + return &containers[i], nil + } + } + return nil, fmt.Errorf("container %q not found in deployment %s/%s", targetName, deployment.Namespace, deployment.Name) +} + +// ownerReference makes the Deployment the controller of a migration Job or +// status ConfigMap so both are garbage collected with it. BlockOwnerDeletion +// is left unset: it needs update on deployments/finalizers, which the +// operator is not granted. +func ownerReference(deployment *appsv1.Deployment) metav1.OwnerReference { + return metav1.OwnerReference{ + APIVersion: "apps/v1", + Kind: "Deployment", + Name: deployment.Name, + UID: deployment.UID, + Controller: ptr.To(true), + } +} + +func isOperatorManagedResourceForDeployment(obj metav1.Object, deployment *appsv1.Deployment) bool { + if obj.GetLabels()[LabelManagedBy] != LabelManagedByValue { + return false + } + for _, owner := range obj.GetOwnerReferences() { + if ptr.Deref(owner.Controller, false) && + owner.APIVersion == "apps/v1" && + owner.Kind == "Deployment" && + owner.Name == deployment.Name { + return true + } + } + return false +} + +func isLegacyMigrationJob(job *batchv1.Job) bool { + return job.Labels[LabelManagedBy] == "Helm" && job.Annotations["helm.sh/hook"] != "" +} + +// buildMigrationJob constructs a Job that runs "openfga migrate" with the +// OpenFGA container's image and environment, the Deployment's pod scheduling, +// and chart-provided migration-specific configuration. Other Deployment +// containers and migration sidecars run as native sidecars. +func (r *MigrationReconciler) buildMigrationJob(deployment *appsv1.Deployment, container *corev1.Container, desired migrationIdentity) (*batchv1.Job, error) { + podSpec := deployment.Spec.Template.Spec + serviceAccount := deployment.Annotations[AnnotationMigrationServiceAccount] + if serviceAccount == "" { + serviceAccount = podSpec.ServiceAccountName + } + + migrationInitContainers, err := annotationJSON[[]corev1.Container](deployment, AnnotationMigrationInitContainers) + if err != nil { + return nil, err + } + migrationSidecars, err := annotationJSON[[]corev1.Container](deployment, AnnotationMigrationSidecars) + if err != nil { + return nil, err + } + extraVolumes, err := annotationJSON[[]corev1.Volume](deployment, AnnotationMigrationVolumes) + if err != nil { + return nil, err + } + extraVolumeMounts, err := annotationJSON[[]corev1.VolumeMount](deployment, AnnotationMigrationVolumeMounts) + if err != nil { + return nil, err + } + migrationAnnotations, err := annotationJSON[map[string]string](deployment, AnnotationMigrationAnnotations) + if err != nil { + return nil, err + } + migrationLabels, err := annotationJSON[map[string]string](deployment, AnnotationMigrationLabels) + if err != nil { + return nil, err + } + + resources := container.Resources + if deployment.Annotations[AnnotationMigrationResources] != "" { + resources, err = annotationJSON[corev1.ResourceRequirements](deployment, AnnotationMigrationResources) + if err != nil { + return nil, err + } + } + env := append([]corev1.EnvVar(nil), container.Env...) + if timeout := deployment.Annotations[AnnotationMigrationTimeout]; timeout != "" && !hasEnvVar(env, "OPENFGA_TIMEOUT") { + env = append(env, corev1.EnvVar{Name: "OPENFGA_TIMEOUT", Value: timeout}) + } + + podAnnotations := mergeStringMaps(migrationAnnotations, map[string]string{ + AnnotationMigrationTrigger: desired.Trigger, + }) + jobLabels := mergeStringMaps(migrationLabels, map[string]string{ + LabelPartOf: LabelPartOfValue, + LabelComponent: "migration", + LabelManagedBy: LabelManagedByValue, + }) + podLabels := mergeStringMaps(migrationLabels, map[string]string{ + LabelPartOf: LabelPartOfValue, + LabelComponent: "migration", + }) + jobAnnotations := mergeStringMaps(migrationAnnotations, map[string]string{ + AnnotationDesiredVersion: desired.Version, + AnnotationMigrationTrigger: desired.Trigger, + }) + + // Native sidecars start before regular init containers and stop when the + // migration container exits. + var runtimeSidecars []corev1.Container + for i := range podSpec.Containers { + if podSpec.Containers[i].Name == container.Name { + continue + } + sidecar := *podSpec.Containers[i].DeepCopy() + runtimeSidecars = append(runtimeSidecars, sidecar) + } + sidecars := mergeContainers(runtimeSidecars, migrationSidecars) + var initContainers []corev1.Container + for i := range sidecars { + sidecar := *sidecars[i].DeepCopy() + sidecar.RestartPolicy = ptr.To(corev1.ContainerRestartPolicyAlways) + initContainers = append(initContainers, sidecar) + } + initContainers = append(initContainers, mergeContainers(podSpec.InitContainers, migrationInitContainers)...) + + containers := []corev1.Container{{ + Name: "migrate-database", + Image: container.Image, + ImagePullPolicy: container.ImagePullPolicy, + Args: []string{"migrate"}, + Env: env, + EnvFrom: container.EnvFrom, + Resources: resources, + VolumeMounts: mergeVolumeMounts(container.VolumeMounts, extraVolumeMounts), + SecurityContext: container.SecurityContext, + }} + + job := &batchv1.Job{ + ObjectMeta: metav1.ObjectMeta{ + Name: migrationJobName(deployment.Name), + Namespace: deployment.Namespace, + Labels: jobLabels, + Annotations: jobAnnotations, + OwnerReferences: []metav1.OwnerReference{ownerReference(deployment)}, + }, + Spec: batchv1.JobSpec{ + BackoffLimit: ptr.To(r.BackoffLimit), + Template: corev1.PodTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{ + Labels: podLabels, + Annotations: podAnnotations, + }, + Spec: corev1.PodSpec{ + ServiceAccountName: serviceAccount, + RestartPolicy: corev1.RestartPolicyNever, + ImagePullSecrets: podSpec.ImagePullSecrets, + SecurityContext: podSpec.SecurityContext, + InitContainers: initContainers, + Containers: containers, + Volumes: mergeVolumes(podSpec.Volumes, extraVolumes), + NodeSelector: podSpec.NodeSelector, + Tolerations: podSpec.Tolerations, + Affinity: podSpec.Affinity, + }, + }, + }, + } + if r.ActiveDeadlineSeconds > 0 { + job.Spec.ActiveDeadlineSeconds = ptr.To(r.ActiveDeadlineSeconds) + } + if job.Annotations == nil { + job.Annotations = map[string]string{} + } + job.Annotations[AnnotationDesiredVersion] = desired.Version + job.Annotations[AnnotationMigrationTrigger] = desired.Trigger + job.Annotations[AnnotationPodTemplateHash] = podTemplateHash(&job.Spec.Template) + return job, nil +} + +func annotationJSON[T any](deployment *appsv1.Deployment, annotation string) (T, error) { + var value T + raw := deployment.Annotations[annotation] + if raw == "" { + return value, nil + } + decoder := json.NewDecoder(strings.NewReader(raw)) + decoder.DisallowUnknownFields() + if err := decoder.Decode(&value); err != nil { + return value, fmt.Errorf("decoding %s annotation on deployment %s/%s: %w", annotation, deployment.Namespace, deployment.Name, err) + } + if err := decoder.Decode(&struct{}{}); err != io.EOF { + return value, fmt.Errorf("decoding %s annotation on deployment %s/%s: trailing JSON data", annotation, deployment.Namespace, deployment.Name) + } + return value, nil +} + +func hasEnvVar(env []corev1.EnvVar, name string) bool { + for i := range env { + if env[i].Name == name { + return true + } + } + return false +} + +func mergeVolumes(base, extra []corev1.Volume) []corev1.Volume { + merged := append([]corev1.Volume(nil), base...) + index := make(map[string]int, len(merged)) + for i := range merged { + index[merged[i].Name] = i + } + for _, volume := range extra { + if i, ok := index[volume.Name]; ok { + merged[i] = volume + continue + } + index[volume.Name] = len(merged) + merged = append(merged, volume) + } + return merged +} + +func mergeContainers(base, overrides []corev1.Container) []corev1.Container { + merged := append([]corev1.Container(nil), base...) + index := make(map[string]int, len(merged)) + for i := range merged { + index[merged[i].Name] = i + } + for _, container := range overrides { + if i, ok := index[container.Name]; ok { + merged[i] = container + continue + } + index[container.Name] = len(merged) + merged = append(merged, container) + } + return merged +} + +func mergeVolumeMounts(base, extra []corev1.VolumeMount) []corev1.VolumeMount { + merged := append([]corev1.VolumeMount(nil), base...) + index := make(map[string]int, len(merged)) + for i := range merged { + index[merged[i].MountPath] = i + } + for _, mount := range extra { + if i, ok := index[mount.MountPath]; ok { + merged[i] = mount + continue + } + index[mount.MountPath] = len(merged) + merged = append(merged, mount) + } + return merged +} + +func mergeStringMaps(base, overrides map[string]string) map[string]string { + merged := make(map[string]string, len(base)+len(overrides)) + for key, value := range base { + merged[key] = value + } + for key, value := range overrides { + merged[key] = value + } + return merged +} + +func podTemplateHash(template *corev1.PodTemplateSpec) string { + b, err := json.Marshal(template) + if err != nil { + panic(err) // a PodTemplateSpec always marshals + } + return fmt.Sprintf("%x", sha256.Sum256(b))[:16] +} + +// updateMigrationStatus records the migrated identity in the status ConfigMap. +func updateMigrationStatus(ctx context.Context, c client.Client, deployment *appsv1.Deployment, identity migrationIdentity, jobName string) error { + cm := &corev1.ConfigMap{ObjectMeta: metav1.ObjectMeta{ + Name: migrationConfigMapName(deployment.Name), + Namespace: deployment.Namespace, + }} + _, err := controllerutil.CreateOrUpdate(ctx, c, cm, func() error { + if cm.ResourceVersion != "" && + !metav1.IsControlledBy(cm, deployment) && + !isOperatorManagedResourceForDeployment(cm, deployment) { + return fmt.Errorf("ConfigMap %s/%s already exists and is not managed by the OpenFGA operator", cm.Namespace, cm.Name) + } + cm.Labels = map[string]string{ + LabelPartOf: LabelPartOfValue, + LabelComponent: "migration", + LabelManagedBy: LabelManagedByValue, + } + cm.OwnerReferences = []metav1.OwnerReference{ownerReference(deployment)} + cm.Data = map[string]string{ + "version": identity.Version, + "trigger": identity.Trigger, + "migratedAt": time.Now().UTC().Format(time.RFC3339), + "jobName": jobName, + } + return nil + }) + if err != nil { + return fmt.Errorf("updating migration status ConfigMap: %w", err) + } + return nil +} diff --git a/operator/internal/controller/migration_controller.go b/operator/internal/controller/migration_controller.go new file mode 100644 index 00000000..e0c0b1b1 --- /dev/null +++ b/operator/internal/controller/migration_controller.go @@ -0,0 +1,341 @@ +package controller + +import ( + "context" + "fmt" + "time" + + appsv1 "k8s.io/api/apps/v1" + batchv1 "k8s.io/api/batch/v1" + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/types" + "k8s.io/client-go/tools/record" + "k8s.io/utils/ptr" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/builder" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/log" + "sigs.k8s.io/controller-runtime/pkg/predicate" +) + +// retryDelay is how long a failed migration Job is kept before it is replaced. +const retryDelay = 60 * time.Second + +// MigrationReconciler watches OpenFGA Deployments and runs a database +// migration Job whenever the OpenFGA image or the datastore trigger changes. +type MigrationReconciler struct { + client.Client + Recorder record.EventRecorder + + // BackoffLimit for migration Jobs. + BackoffLimit int32 + // ActiveDeadlineSeconds for migration Jobs; 0 means no deadline. + ActiveDeadlineSeconds int64 + // TTLSecondsAfterFinished is applied to migration Jobs once they succeed. + TTLSecondsAfterFinished int32 +} + +// Reconcile handles a single reconciliation for an OpenFGA Deployment. +func (r *MigrationReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) { + logger := log.FromContext(ctx) + + deployment := &appsv1.Deployment{} + if err := r.Get(ctx, req.NamespacedName, deployment); err != nil { + return ctrl.Result{}, client.IgnoreNotFound(err) + } + if deployment.Annotations[AnnotationMigrationEnabled] != "true" { + return ctrl.Result{}, nil + } + + container, err := findOpenFGAContainer(deployment) + if err != nil { + return ctrl.Result{}, err + } + desired := desiredIdentity(deployment, container) + desiredJob, err := r.buildMigrationJob(deployment, container, desired) + if err != nil { + return ctrl.Result{}, err + } + + status := &corev1.ConfigMap{} + err = r.Get(ctx, types.NamespacedName{Name: migrationConfigMapName(req.Name), Namespace: req.Namespace}, status) + if err != nil && !apierrors.IsNotFound(err) { + return ctrl.Result{}, fmt.Errorf("getting migration status: %w", err) + } + statusOwnedByDeployment := err == nil && metav1.IsControlledBy(status, deployment) + if err == nil && !statusOwnedByDeployment && !isOperatorManagedResourceForDeployment(status, deployment) { + return ctrl.Result{}, fmt.Errorf("migration status ConfigMap %s/%s already exists and is not managed by this Deployment", status.Namespace, status.Name) + } + current := recordedIdentity(status, deployment) + + job := &batchv1.Job{} + err = r.Get(ctx, types.NamespacedName{Name: migrationJobName(req.Name), Namespace: req.Namespace}, job) + if err != nil && !apierrors.IsNotFound(err) { + return ctrl.Result{}, fmt.Errorf("getting migration job: %w", err) + } + if err != nil { + job = nil + } + + if current == desired { + if job != nil && metav1.IsControlledBy(job, deployment) && isJobConditionTrue(job, batchv1.JobComplete) { + if err := r.setCompletedJobTTL(ctx, job); err != nil { + return ctrl.Result{}, err + } + } + _, err := r.patchCondition(ctx, deployment, clearMigrationFailedCondition) + return ctrl.Result{}, err + } + + if job == nil { + job = desiredJob + if err := r.Create(ctx, job); err != nil { + if apierrors.IsAlreadyExists(err) { + // The cache has not caught up with a Job created by an earlier reconcile. + return ctrl.Result{RequeueAfter: 5 * time.Second}, nil + } + return ctrl.Result{}, fmt.Errorf("creating migration job: %w", err) + } + logger.Info("created migration job", "job", job.Name, "currentVersion", current.Version, "desiredVersion", desired.Version) + r.Recorder.Eventf(deployment, corev1.EventTypeNormal, "MigrationStarted", "Created migration job %s for version %s", job.Name, desired.Version) + return ctrl.Result{RequeueAfter: 10 * time.Second}, nil + } + + if job.DeletionTimestamp != nil { + // Foreground deletion in progress: the name frees up once the pods are gone. + return ctrl.Result{RequeueAfter: 5 * time.Second}, nil + } + jobOwnedByDeployment := metav1.IsControlledBy(job, deployment) + replaceableJob := jobOwnedByDeployment || isOperatorManagedResourceForDeployment(job, deployment) || isLegacyMigrationJob(job) + if !replaceableJob { + r.Recorder.Eventf(deployment, corev1.EventTypeWarning, "MigrationJobConflict", "Job %s exists but was not created by the operator or the chart; delete or rename it to let the migration run", job.Name) + return ctrl.Result{}, fmt.Errorf("migration Job %s/%s already exists and is not managed by this Deployment", job.Namespace, job.Name) + } + + // Only a Job this operator created for the desired identity is trusted. + // Anything else under the same name, such as a Job for a previous image or + // the chart's legacy Helm hook Job, is replaced. + complete := isJobConditionTrue(job, batchv1.JobComplete) + failedAt, failed := jobFailedAt(job) + // A pod that is Ready is running the migration; one that already succeeded + // finished before the Job condition was written. A pod that merely exists + // (Active) may be stuck in Pending and has not started anything. + started := !complete && !failed && (ptr.Deref(job.Status.Ready, 0) > 0 || job.Status.Succeeded > 0) + outdated := !jobOwnedByDeployment || jobIdentity(job) != desired + // A Job whose pod cannot start (a bad secret reference, an image pull + // error, an unschedulable pod) never fails on its own, so rebuild it once + // the Deployment's pod template has changed. + if !outdated && !complete && !failed && !started { + outdated = job.Annotations[AnnotationPodTemplateHash] != desiredJob.Annotations[AnnotationPodTemplateHash] + } + if outdated { + // Never interrupt a running migration: a non-transactional step such as + // a concurrent index build that is aborted halfway leaves the schema in + // a state the next run does not repair. A pod that finished before the + // Job condition was written is also left alone. + if started { + logger.V(1).Info("waiting for started migration job before replacing it", "job", job.Name, "jobVersion", jobIdentity(job).Version, "desiredVersion", desired.Version) + return ctrl.Result{RequeueAfter: 10 * time.Second}, nil + } + logger.Info("replacing migration job", "job", job.Name, "jobVersion", jobIdentity(job).Version, "desiredVersion", desired.Version) + if err := r.deleteJob(ctx, job); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{RequeueAfter: 5 * time.Second}, nil + } + + if complete { + if err := updateMigrationStatus(ctx, r.Client, deployment, desired, job.Name); err != nil { + return ctrl.Result{}, err + } + logger.Info("migration succeeded", "version", desired.Version) + r.Recorder.Eventf(deployment, corev1.EventTypeNormal, "MigrationSucceeded", "Database migrated to version %s", desired.Version) + // Only now: a TTL of 0 would otherwise remove the Job before its + // outcome is recorded, and a new Job would run the migration again. + if err := r.setCompletedJobTTL(ctx, job); err != nil { + return ctrl.Result{}, err + } + _, err := r.patchCondition(ctx, deployment, clearMigrationFailedCondition) + return ctrl.Result{}, err + } + + if failed { + changed, err := r.patchCondition(ctx, deployment, func(d *appsv1.Deployment) bool { + return setMigrationFailedCondition(d, desired.Version) + }) + if err != nil { + return ctrl.Result{}, err + } + if changed { + logger.Info("migration job failed", "job", job.Name, "version", desired.Version, "retryIn", retryDelay) + r.Recorder.Eventf(deployment, corev1.EventTypeWarning, "MigrationFailed", "Migration job %s failed for version %s; retrying in %s", job.Name, desired.Version, retryDelay) + } + // The failed Job itself is the retry timer, so the delay survives + // operator restarts and leaves the pod logs around to inspect. + if wait := retryDelay - time.Since(failedAt); wait > 0 { + return ctrl.Result{RequeueAfter: wait}, nil + } + logger.Info("retrying migration", "job", job.Name, "version", desired.Version) + if err := r.deleteJob(ctx, job); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{RequeueAfter: 5 * time.Second}, nil + } + + return ctrl.Result{RequeueAfter: 10 * time.Second}, nil +} + +// setCompletedJobTTL applies cleanup only after migration status is durable. +// Optimistic locking avoids overwriting a concurrent Job spec update. +func (r *MigrationReconciler) setCompletedJobTTL(ctx context.Context, job *batchv1.Job) error { + if job.Spec.TTLSecondsAfterFinished != nil && *job.Spec.TTLSecondsAfterFinished == r.TTLSecondsAfterFinished { + return nil + } + patch := client.MergeFromWithOptions(job.DeepCopy(), client.MergeFromWithOptimisticLock{}) + job.Spec.TTLSecondsAfterFinished = ptr.To(r.TTLSecondsAfterFinished) + if err := r.Patch(ctx, job, patch); err != nil { + return fmt.Errorf("setting completed migration Job TTL: %w", err) + } + return nil +} + +// deleteJob removes a migration Job and waits for its pods to be gone before +// the name is reused, so two migrations never run at once. Preconditions make +// sure the Job is still the one that was inspected. +func (r *MigrationReconciler) deleteJob(ctx context.Context, job *batchv1.Job) error { + options := []client.DeleteOption{client.PropagationPolicy(metav1.DeletePropagationForeground)} + preconditions := client.Preconditions{} + hasPreconditions := false + if job.UID != "" { + uid := job.UID + preconditions.UID = &uid + hasPreconditions = true + } + if job.ResourceVersion != "" { + resourceVersion := job.ResourceVersion + preconditions.ResourceVersion = &resourceVersion + hasPreconditions = true + } + if hasPreconditions { + options = append(options, preconditions) + } + err := r.Delete(ctx, job, options...) + if apierrors.IsNotFound(err) || apierrors.IsConflict(err) { + // Gone already, or changed since it was read: the requeue re-evaluates it. + return nil + } + if err != nil { + return fmt.Errorf("deleting migration job %s: %w", job.Name, err) + } + return nil +} + +// patchCondition applies update to the Deployment's status conditions and +// patches the status only if something changed. The strategic merge patch +// merges conditions by type, so the Deployment controller's own conditions are +// left alone. +func (r *MigrationReconciler) patchCondition(ctx context.Context, deployment *appsv1.Deployment, update func(*appsv1.Deployment) bool) (bool, error) { + patch := client.StrategicMergeFrom(deployment.DeepCopy(), client.MergeFromWithOptimisticLock{}) + if !update(deployment) { + return false, nil + } + if err := r.Status().Patch(ctx, deployment, patch); err != nil { + return false, fmt.Errorf("patching MigrationFailed condition: %w", err) + } + return true, nil +} + +func isJobConditionTrue(job *batchv1.Job, conditionType batchv1.JobConditionType) bool { + for _, c := range job.Status.Conditions { + if c.Type == conditionType && c.Status == corev1.ConditionTrue { + return true + } + } + return false +} + +// jobFailedAt reports whether the Job has failed and when. JobFailureTarget is +// set as soon as the Job controller decides the Job will fail; JobFailed only +// once its pods have terminated. +func jobFailedAt(job *batchv1.Job) (time.Time, bool) { + for _, c := range job.Status.Conditions { + if (c.Type == batchv1.JobFailureTarget || c.Type == batchv1.JobFailed) && c.Status == corev1.ConditionTrue { + return c.LastTransitionTime.Time, true + } + } + return time.Time{}, false +} + +// setMigrationFailedCondition sets a MigrationFailed condition on the Deployment +// and reports whether anything changed. LastTransitionTime only advances on a +// real status transition. +// +// Deployment.Status.Conditions is []appsv1.DeploymentCondition rather than +// []metav1.Condition, so meta.SetStatusCondition cannot be used. +func setMigrationFailedCondition(deployment *appsv1.Deployment, version string) bool { + message := fmt.Sprintf("Database migration failed for version %s. Check migration job logs.", version) + for i, c := range deployment.Status.Conditions { + if c.Type == "MigrationFailed" { + if c.Status == corev1.ConditionTrue && c.Reason == "MigrationJobFailed" && c.Message == message { + return false + } + if c.Status != corev1.ConditionTrue { + deployment.Status.Conditions[i].LastTransitionTime = metav1.Now() + } + deployment.Status.Conditions[i].Status = corev1.ConditionTrue + deployment.Status.Conditions[i].Reason = "MigrationJobFailed" + deployment.Status.Conditions[i].Message = message + return true + } + } + deployment.Status.Conditions = append(deployment.Status.Conditions, appsv1.DeploymentCondition{ + Type: "MigrationFailed", + Status: corev1.ConditionTrue, + LastTransitionTime: metav1.Now(), + Reason: "MigrationJobFailed", + Message: message, + }) + return true +} + +// clearMigrationFailedCondition sets an existing MigrationFailed condition to +// False and reports whether anything changed. It is a no-op when the condition +// is absent or already False, so healthy Deployments are never re-patched. +func clearMigrationFailedCondition(deployment *appsv1.Deployment) bool { + for i, c := range deployment.Status.Conditions { + if c.Type == "MigrationFailed" { + if c.Status == corev1.ConditionFalse { + return false + } + deployment.Status.Conditions[i].Status = corev1.ConditionFalse + deployment.Status.Conditions[i].LastTransitionTime = metav1.Now() + deployment.Status.Conditions[i].Reason = "MigrationSucceeded" + deployment.Status.Conditions[i].Message = "Migration completed successfully." + return true + } + } + return false +} + +// SetupWithManager sets up the controller with the Manager. +func (r *MigrationReconciler) SetupWithManager(mgr ctrl.Manager) error { + openfgaDeployments, err := predicate.LabelSelectorPredicate(metav1.LabelSelector{ + MatchLabels: map[string]string{ + LabelPartOf: LabelPartOfValue, + LabelComponent: LabelComponentValue, + }, + }) + if err != nil { + return fmt.Errorf("creating label predicate: %w", err) + } + + // Owning the status ConfigMap means deleting it triggers a reconcile, which + // runs the migration again once the previous Job is gone. + return ctrl.NewControllerManagedBy(mgr). + For(&appsv1.Deployment{}, builder.WithPredicates(openfgaDeployments)). + Owns(&batchv1.Job{}). + Owns(&corev1.ConfigMap{}). + Complete(r) +} diff --git a/operator/internal/controller/migration_controller_test.go b/operator/internal/controller/migration_controller_test.go new file mode 100644 index 00000000..0f10d3d0 --- /dev/null +++ b/operator/internal/controller/migration_controller_test.go @@ -0,0 +1,1070 @@ +package controller + +import ( + "context" + "fmt" + "strings" + "testing" + "time" + + appsv1 "k8s.io/api/apps/v1" + batchv1 "k8s.io/api/batch/v1" + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/types" + clientgoscheme "k8s.io/client-go/kubernetes/scheme" + "k8s.io/client-go/tools/record" + "k8s.io/utils/ptr" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/client/fake" + "sigs.k8s.io/controller-runtime/pkg/client/interceptor" +) + +var ( + deploymentKey = types.NamespacedName{Name: "openfga", Namespace: "default"} + jobKey = types.NamespacedName{Name: "openfga-migrate", Namespace: "default"} + statusKey = types.NamespacedName{Name: "openfga-migration-status", Namespace: "default"} +) + +func newTestDeployment(image string) *appsv1.Deployment { + return &appsv1.Deployment{ + ObjectMeta: metav1.ObjectMeta{ + Name: deploymentKey.Name, + Namespace: deploymentKey.Namespace, + UID: "test-uid-123", + Labels: map[string]string{ + LabelPartOf: LabelPartOfValue, + LabelComponent: LabelComponentValue, + }, + Annotations: map[string]string{ + AnnotationMigrationEnabled: "true", + }, + }, + Spec: appsv1.DeploymentSpec{ + Replicas: ptr.To(int32(3)), + Selector: &metav1.LabelSelector{MatchLabels: map[string]string{"app": "openfga"}}, + Template: corev1.PodTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{Labels: map[string]string{"app": "openfga"}}, + Spec: corev1.PodSpec{ + ServiceAccountName: "openfga", + Containers: []corev1.Container{{ + Name: "openfga", + Image: image, + Env: []corev1.EnvVar{ + {Name: "OPENFGA_DATASTORE_ENGINE", Value: "postgres"}, + {Name: "OPENFGA_DATASTORE_URI", Value: "postgres://localhost/openfga"}, + {Name: "OPENFGA_LOG_LEVEL", Value: "info"}, + }, + }}, + }, + }, + }, + } +} + +// newTestJob returns the migration Job the operator builds for dep. +func newTestJob(dep *appsv1.Deployment, conditions ...batchv1.JobCondition) *batchv1.Job { + container := &dep.Spec.Template.Spec.Containers[0] + job, err := (&MigrationReconciler{TTLSecondsAfterFinished: DefaultTTLSecondsAfterFinished}).buildMigrationJob(dep, container, desiredIdentity(dep, container)) + if err != nil { + panic(err) + } + job.Status.Conditions = conditions + return job +} + +func jobCondition(t batchv1.JobConditionType, at time.Time) batchv1.JobCondition { + return batchv1.JobCondition{Type: t, Status: corev1.ConditionTrue, LastTransitionTime: metav1.NewTime(at)} +} + +func newStatus(dep *appsv1.Deployment) *corev1.ConfigMap { + job := newTestJob(dep) + return &corev1.ConfigMap{ + ObjectMeta: metav1.ObjectMeta{ + Name: statusKey.Name, + Namespace: statusKey.Namespace, + Labels: map[string]string{LabelManagedBy: LabelManagedByValue}, + OwnerReferences: []metav1.OwnerReference{ + ownerReference(dep), + }, + }, + Data: map[string]string{ + "version": job.Annotations[AnnotationDesiredVersion], + "trigger": job.Annotations[AnnotationMigrationTrigger], + }, + } +} + +func newReconciler(t *testing.T, funcs *interceptor.Funcs, objects ...client.Object) *MigrationReconciler { + t.Helper() + scheme := runtime.NewScheme() + if err := clientgoscheme.AddToScheme(scheme); err != nil { + t.Fatal(err) + } + b := fake.NewClientBuilder().WithScheme(scheme). + WithStatusSubresource(&appsv1.Deployment{}). + WithObjects(objects...) + if funcs != nil { + b = b.WithInterceptorFuncs(*funcs) + } + return &MigrationReconciler{ + Client: b.Build(), + Recorder: record.NewFakeRecorder(20), + BackoffLimit: DefaultBackoffLimit, + ActiveDeadlineSeconds: DefaultActiveDeadlineSeconds, + TTLSecondsAfterFinished: DefaultTTLSecondsAfterFinished, + } +} + +func events(r *MigrationReconciler) []string { + var out []string + for { + select { + case e := <-r.Recorder.(*record.FakeRecorder).Events: + out = append(out, e) + default: + return out + } + } +} + +var failStatusPatch = &interceptor.Funcs{ + SubResourcePatch: func(ctx context.Context, c client.Client, subResource string, obj client.Object, patch client.Patch, opts ...client.SubResourcePatchOption) error { + if subResource == "status" { + return fmt.Errorf("simulated status patch error") + } + return c.SubResource(subResource).Patch(ctx, obj, patch, opts...) + }, +} + +func reconcileOnce(t *testing.T, r *MigrationReconciler) ctrl.Result { + t.Helper() + result, err := r.Reconcile(context.Background(), ctrl.Request{NamespacedName: deploymentKey}) + if err != nil { + t.Fatalf("unexpected reconcile error: %v", err) + } + return result +} + +func getDeployment(t *testing.T, r *MigrationReconciler) *appsv1.Deployment { + t.Helper() + d := &appsv1.Deployment{} + if err := r.Get(context.Background(), deploymentKey, d); err != nil { + t.Fatalf("getting deployment: %v", err) + } + return d +} + +func getJob(r *MigrationReconciler) (*batchv1.Job, error) { + job := &batchv1.Job{} + return job, r.Get(context.Background(), jobKey, job) +} + +func getStatus(r *MigrationReconciler) (*corev1.ConfigMap, error) { + cm := &corev1.ConfigMap{} + return cm, r.Get(context.Background(), statusKey, cm) +} + +func findCondition(conditions []appsv1.DeploymentCondition, condType string) *appsv1.DeploymentCondition { + for i := range conditions { + if string(conditions[i].Type) == condType { + return &conditions[i] + } + } + return nil +} + +func TestReconcile_FirstInstall_CreatesJob(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + dep.Annotations[AnnotationMigrationServiceAccount] = "openfga-migration" + dep.Spec.Template.Spec.Containers[0].ImagePullPolicy = corev1.PullAlways + r := newReconciler(t, nil, dep) + + if result := reconcileOnce(t, r); result.RequeueAfter == 0 { + t.Error("expected a requeue to poll the new Job") + } + + job, err := getJob(r) + if err != nil { + t.Fatalf("expected migration job to be created: %v", err) + } + c := job.Spec.Template.Spec.Containers[0] + if c.Image != "openfga/openfga:v1.14.0" || len(c.Args) != 1 || c.Args[0] != "migrate" { + t.Errorf("unexpected migrate container: image=%s args=%v", c.Image, c.Args) + } + if c.ImagePullPolicy != corev1.PullAlways { + t.Errorf("expected the OpenFGA container's pull policy, got %q", c.ImagePullPolicy) + } + if len(c.Env) != 3 { + t.Errorf("expected all OpenFGA env vars to be passed through, got %v", c.Env) + } + if sa := job.Spec.Template.Spec.ServiceAccountName; sa != "openfga-migration" { + t.Errorf("expected migration service account, got %q", sa) + } + if got := job.Annotations[AnnotationDesiredVersion]; got != "v1.14.0" { + t.Errorf("expected desired-version annotation v1.14.0, got %q", got) + } + if job.Spec.ActiveDeadlineSeconds != nil { + t.Errorf("expected no deadline by default, got %d", *job.Spec.ActiveDeadlineSeconds) + } + if job.Spec.TTLSecondsAfterFinished != nil { + t.Errorf("expected TTL to remain unset until the Job succeeds, got %d", *job.Spec.TTLSecondsAfterFinished) + } + if len(job.Spec.Template.Spec.InitContainers) != 0 { + t.Errorf("expected no sidecars for a single-container deployment, got %d", len(job.Spec.Template.Spec.InitContainers)) + } + if len(job.OwnerReferences) != 1 || !ptr.Deref(job.OwnerReferences[0].Controller, false) || job.OwnerReferences[0].BlockOwnerDeletion != nil { + t.Errorf("expected a single controller owner reference without blockOwnerDeletion, got %+v", job.OwnerReferences) + } + if replicas := *getDeployment(t, r).Spec.Replicas; replicas != 3 { + t.Errorf("replicas must not be changed, got %d", replicas) + } +} + +func TestReconcile_FirstInstall_DeadlineAndDefaultServiceAccount(t *testing.T) { + r := newReconciler(t, nil, newTestDeployment("openfga/openfga:v1.14.0")) + r.ActiveDeadlineSeconds = 600 + reconcileOnce(t, r) + + job, err := getJob(r) + if err != nil { + t.Fatalf("expected migration job to be created: %v", err) + } + if got := ptr.Deref(job.Spec.ActiveDeadlineSeconds, 0); got != 600 { + t.Errorf("expected activeDeadlineSeconds 600, got %d", got) + } + if sa := job.Spec.Template.Spec.ServiceAccountName; sa != "openfga" { + t.Errorf("expected the Deployment's service account, got %q", sa) + } +} + +func TestReconcile_JobAlreadyExistsOnCreate_Requeues(t *testing.T) { + r := newReconciler(t, &interceptor.Funcs{ + Create: func(ctx context.Context, c client.WithWatch, obj client.Object, opts ...client.CreateOption) error { + return apierrors.NewAlreadyExists(batchv1.Resource("jobs"), obj.GetName()) + }, + }, newTestDeployment("openfga/openfga:v1.14.0")) + + if result := reconcileOnce(t, r); result.RequeueAfter == 0 { + t.Error("expected a requeue when the Job already exists") + } +} + +func TestReconcile_VersionMatch_ClearsFailedCondition(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + dep.Status.Conditions = []appsv1.DeploymentCondition{{Type: "MigrationFailed", Status: corev1.ConditionTrue}} + r := newReconciler(t, nil, dep, newStatus(dep)) + + if result := reconcileOnce(t, r); result.RequeueAfter != 0 { + t.Errorf("expected no requeue when versions match, got %v", result.RequeueAfter) + } + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Errorf("expected no migration job, got err=%v", err) + } + updated := getDeployment(t, r) + if cond := findCondition(updated.Status.Conditions, "MigrationFailed"); cond == nil || cond.Status != corev1.ConditionFalse { + t.Errorf("expected MigrationFailed=False, got %+v", cond) + } + if *updated.Spec.Replicas != 3 { + t.Errorf("replicas must not be changed, got %d", *updated.Spec.Replicas) + } +} + +func TestReconcile_VersionMatch_StatusPatchError(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + dep.Status.Conditions = []appsv1.DeploymentCondition{{Type: "MigrationFailed", Status: corev1.ConditionTrue}} + r := newReconciler(t, failStatusPatch, dep, newStatus(dep)) + + if _, err := r.Reconcile(context.Background(), ctrl.Request{NamespacedName: deploymentKey}); err == nil { + t.Fatal("expected the status patch error to be returned") + } +} + +func TestReconcile_JobSucceeded_CreatesStatus(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + dep.Status.Conditions = []appsv1.DeploymentCondition{{Type: "MigrationFailed", Status: corev1.ConditionTrue, Reason: "MigrationJobFailed"}} + job := newTestJob(dep, jobCondition(batchv1.JobComplete, time.Now())) + r := newReconciler(t, nil, dep, job) + + if result := reconcileOnce(t, r); result.RequeueAfter != 0 { + t.Errorf("expected no requeue after success, got %v", result.RequeueAfter) + } + cm, err := getStatus(r) + if err != nil { + t.Fatalf("expected migration status ConfigMap: %v", err) + } + if cm.Data["version"] != "v1.14.0" || cm.Data["jobName"] != jobKey.Name { + t.Errorf("unexpected status data: %v", cm.Data) + } + if len(cm.OwnerReferences) != 1 || cm.OwnerReferences[0].UID != "test-uid-123" { + t.Errorf("expected the Deployment to own the status ConfigMap, got %+v", cm.OwnerReferences) + } + completedJob, err := getJob(r) + if err != nil { + t.Fatalf("expected the completed Job: %v", err) + } + if got := ptr.Deref(completedJob.Spec.TTLSecondsAfterFinished, -1); got != DefaultTTLSecondsAfterFinished { + t.Errorf("expected completed Job TTL %d, got %d", DefaultTTLSecondsAfterFinished, got) + } + cond := findCondition(getDeployment(t, r).Status.Conditions, "MigrationFailed") + if cond == nil || cond.Status != corev1.ConditionFalse || cond.Reason != "MigrationSucceeded" { + t.Errorf("expected MigrationFailed=False/MigrationSucceeded, got %+v", cond) + } +} + +func TestReconcile_TTLZero_AppliedOnlyAfterSuccess(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + failedJob := newTestJob(dep, jobCondition(batchv1.JobFailed, time.Now())) + r := newReconciler(t, nil, dep, failedJob) + r.TTLSecondsAfterFinished = 0 + + reconcileOnce(t, r) + job, err := getJob(r) + if err != nil { + t.Fatalf("failed job must be kept for the retry delay: %v", err) + } + if job.Spec.TTLSecondsAfterFinished != nil { + t.Errorf("a failed job must not get a TTL, got %d", *job.Spec.TTLSecondsAfterFinished) + } + + job.Status = batchv1.JobStatus{Succeeded: 1, Conditions: []batchv1.JobCondition{jobCondition(batchv1.JobComplete, time.Now())}} + if err := r.Status().Update(context.Background(), job); err != nil { + t.Fatal(err) + } + reconcileOnce(t, r) + if _, err := getStatus(r); err != nil { + t.Fatalf("status must be recorded before the job can be garbage collected: %v", err) + } + job, _ = getJob(r) + if got := ptr.Deref(job.Spec.TTLSecondsAfterFinished, -1); got != 0 { + t.Errorf("expected TTL 0 after success, got %d", got) + } +} + +func TestReconcile_UpToDate_AppliesMissingTTL(t *testing.T) { + // The TTL patch failed after the status was recorded; the up-to-date path + // must finish the job off so it does not linger forever. + dep := newTestDeployment("openfga/openfga:v1.14.0") + r := newReconciler(t, nil, dep, newStatus(dep), newTestJob(dep, jobCondition(batchv1.JobComplete, time.Now()))) + + reconcileOnce(t, r) + job, err := getJob(r) + if err != nil { + t.Fatal(err) + } + if got := ptr.Deref(job.Spec.TTLSecondsAfterFinished, -1); got != DefaultTTLSecondsAfterFinished { + t.Errorf("expected TTL %d, got %d", DefaultTTLSecondsAfterFinished, got) + } +} + +func TestReconcile_JobSucceeded_AppliesZeroTTL(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + r := newReconciler(t, nil, dep, newTestJob(dep, jobCondition(batchv1.JobComplete, time.Now()))) + r.TTLSecondsAfterFinished = 0 + + reconcileOnce(t, r) + + job, err := getJob(r) + if err != nil { + t.Fatalf("expected the fake client to retain the completed Job: %v", err) + } + if job.Spec.TTLSecondsAfterFinished == nil || *job.Spec.TTLSecondsAfterFinished != 0 { + t.Errorf("expected completed Job TTL 0, got %v", job.Spec.TTLSecondsAfterFinished) + } +} + +func TestReconcile_JobSucceeded_UpdatesStatus(t *testing.T) { + oldDep := newTestDeployment("openfga/openfga:v1.13.0") + oldDep.UID = "old-uid" + status := newStatus(oldDep) + dep := newTestDeployment("openfga/openfga:v1.14.0") + job := newTestJob(dep, jobCondition(batchv1.JobComplete, time.Now())) + r := newReconciler(t, nil, dep, status, job) + + reconcileOnce(t, r) + + cm, err := getStatus(r) + if err != nil { + t.Fatalf("expected migration status ConfigMap: %v", err) + } + if cm.Data["version"] != "v1.14.0" { + t.Errorf("expected version v1.14.0, got %q", cm.Data["version"]) + } + if cm.OwnerReferences[0].UID != "test-uid-123" { + t.Errorf("expected owner reference to be reset to the current Deployment, got %+v", cm.OwnerReferences) + } +} + +func TestReconcile_JobInProgress_Requeues(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + r := newReconciler(t, nil, dep, newTestJob(dep)) + + if result := reconcileOnce(t, r); result.RequeueAfter != 10*time.Second { + t.Errorf("expected 10s requeue for an in-progress job, got %v", result.RequeueAfter) + } + if _, err := getStatus(r); !apierrors.IsNotFound(err) { + t.Errorf("expected no status ConfigMap while the job runs, got err=%v", err) + } +} + +func TestReconcile_JobFailed_KeepsJobUntilRetryDelay(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + r := newReconciler(t, nil, dep, newTestJob(dep, jobCondition(batchv1.JobFailed, time.Now().Add(-10*time.Second)))) + + result := reconcileOnce(t, r) + if result.RequeueAfter <= 0 || result.RequeueAfter > retryDelay-10*time.Second { + t.Errorf("expected a requeue for the rest of the retry delay, got %v", result.RequeueAfter) + } + if _, err := getJob(r); err != nil { + t.Errorf("expected the failed job to be kept during the retry delay: %v", err) + } + job, err := getJob(r) + if err != nil { + t.Fatalf("expected the failed Job: %v", err) + } + if job.Spec.TTLSecondsAfterFinished != nil { + t.Errorf("failed Jobs must not have a completion TTL, got %d", *job.Spec.TTLSecondsAfterFinished) + } + cond := findCondition(getDeployment(t, r).Status.Conditions, "MigrationFailed") + if cond == nil || cond.Status != corev1.ConditionTrue || cond.Reason != "MigrationJobFailed" { + t.Errorf("expected MigrationFailed=True, got %+v", cond) + } +} + +func TestReconcile_JobFailed_RetriesAfterDelay(t *testing.T) { + for _, condType := range []batchv1.JobConditionType{batchv1.JobFailed, batchv1.JobFailureTarget} { + t.Run(string(condType), func(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + r := newReconciler(t, nil, dep, newTestJob(dep, jobCondition(condType, time.Now().Add(-2*retryDelay)))) + + reconcileOnce(t, r) + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Fatalf("expected the failed job to be deleted, got err=%v", err) + } + if cond := findCondition(getDeployment(t, r).Status.Conditions, "MigrationFailed"); cond == nil || cond.Status != corev1.ConditionTrue { + t.Errorf("expected MigrationFailed=True, got %+v", cond) + } + + reconcileOnce(t, r) + job, err := getJob(r) + if err != nil { + t.Fatalf("expected a new migration job: %v", err) + } + if len(job.Status.Conditions) != 0 { + t.Errorf("expected a fresh job, got conditions %+v", job.Status.Conditions) + } + }) + } +} + +func TestReconcile_JobFailed_StatusPatchErrorKeepsJob(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + r := newReconciler(t, failStatusPatch, dep, newTestJob(dep, jobCondition(batchv1.JobFailed, time.Now().Add(-2*retryDelay)))) + + if _, err := r.Reconcile(context.Background(), ctrl.Request{NamespacedName: deploymentKey}); err == nil { + t.Fatal("expected the status patch error to be returned") + } + if _, err := getJob(r); err != nil { + t.Errorf("the failed job must be kept until the failure is recorded: %v", err) + } +} + +func TestReconcile_JobForOtherVersion_Replaced(t *testing.T) { + stale := newTestJob(newTestDeployment("openfga/openfga:v1.14.0"), jobCondition(batchv1.JobComplete, time.Now())) + var deleteOpts client.DeleteOptions + r := newReconciler(t, &interceptor.Funcs{ + Delete: func(ctx context.Context, c client.WithWatch, obj client.Object, opts ...client.DeleteOption) error { + deleteOpts.ApplyOptions(opts) + return c.Delete(ctx, obj, opts...) + }, + }, newTestDeployment("openfga/openfga:v1.15.0"), stale) + + if result := reconcileOnce(t, r); result.RequeueAfter == 0 { + t.Error("expected a requeue after deleting the stale job") + } + // Foreground deletion keeps the name taken until the old pods are gone, so + // two migrations never overlap; the preconditions pin the inspected object. + if ptr.Deref(deleteOpts.PropagationPolicy, "") != metav1.DeletePropagationForeground { + t.Errorf("expected foreground deletion, got %v", deleteOpts.PropagationPolicy) + } + if deleteOpts.Preconditions == nil || ptr.Deref(deleteOpts.Preconditions.UID, "") != stale.UID || deleteOpts.Preconditions.ResourceVersion == nil { + t.Errorf("expected UID and resourceVersion preconditions, got %+v", deleteOpts.Preconditions) + } + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Errorf("expected the stale job to be deleted, got err=%v", err) + } + if _, err := getStatus(r); !apierrors.IsNotFound(err) { + t.Errorf("a job for another version must not be recorded as migrated, got err=%v", err) + } +} + +func TestReconcile_UnstartedJobWithOutdatedTemplate_Replaced(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + job := newTestJob(dep) + // The pod exists but cannot start, e.g. CreateContainerConfigError. + job.Status.Active = 1 + job.Status.Ready = ptr.To(int32(0)) + dep.Spec.Template.Spec.Containers[0].Env[1].Value = "postgres://db.example.com/openfga" + r := newReconciler(t, nil, dep, job) + + reconcileOnce(t, r) + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Fatalf("expected the job built from the old pod template to be deleted, got err=%v", err) + } + + reconcileOnce(t, r) + rebuilt, err := getJob(r) + if err != nil { + t.Fatalf("expected a new migration job: %v", err) + } + if uri := rebuilt.Spec.Template.Spec.Containers[0].Env[1].Value; uri != "postgres://db.example.com/openfga" { + t.Errorf("expected the new job to use the updated env, got %q", uri) + } +} + +func TestReconcile_StatusWithOtherTrigger_Reruns(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + status := newStatus(dep) + dep.Annotations[AnnotationMigrationTrigger] = "secret-rotation-2" + r := newReconciler(t, nil, dep, status) + + reconcileOnce(t, r) + job, err := getJob(r) + if err != nil { + t.Fatalf("expected a changed trigger to create a Job: %v", err) + } + if job.Annotations[AnnotationMigrationTrigger] != "secret-rotation-2" { + t.Errorf("expected the Job to carry the new trigger, got %q", job.Annotations[AnnotationMigrationTrigger]) + } +} + +func TestReconcile_UpToDate_IgnoresPodTemplateChanges(t *testing.T) { + // The recorded identity is the image and the trigger. A change to the pod + // template alone, such as a log level or resource limits, is not a reason + // to run the migration again. + dep := newTestDeployment("openfga/openfga:v1.14.0") + status := newStatus(dep) + dep.Spec.Template.Spec.Containers[0].Env[2].Value = "debug" + r := newReconciler(t, nil, dep, status) + + if result := reconcileOnce(t, r); result.RequeueAfter != 0 { + t.Errorf("expected no requeue for an up-to-date migration, got %v", result.RequeueAfter) + } + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Errorf("a pod template change must not run a migration, got err=%v", err) + } +} + +func TestReconcile_VersionOnlyStatus(t *testing.T) { + // A status written before the trigger existed has no trigger key. It still + // matches a Deployment without a trigger, and mismatches one with a trigger. + versionOnly := func(dep *appsv1.Deployment) *corev1.ConfigMap { + return &corev1.ConfigMap{ + ObjectMeta: metav1.ObjectMeta{ + Name: statusKey.Name, + Namespace: statusKey.Namespace, + Labels: map[string]string{LabelManagedBy: LabelManagedByValue}, + OwnerReferences: []metav1.OwnerReference{ownerReference(dep)}, + }, + Data: map[string]string{"version": "v1.14.0"}, + } + } + + dep := newTestDeployment("openfga/openfga:v1.14.0") + r := newReconciler(t, nil, dep, versionOnly(dep)) + reconcileOnce(t, r) + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Errorf("no trigger on either side must not run a migration, got err=%v", err) + } + + dep = newTestDeployment("openfga/openfga:v1.14.0") + dep.Annotations[AnnotationMigrationTrigger] = "abc" + r = newReconciler(t, nil, dep, versionOnly(dep)) + reconcileOnce(t, r) + if _, err := getJob(r); err != nil { + t.Errorf("a Deployment with a trigger must migrate over a version-only status: %v", err) + } +} + +func TestReconcile_StatusOwnedByPreviousDeployment_Reruns(t *testing.T) { + oldDep := newTestDeployment("openfga/openfga:v1.14.0") + oldDep.UID = "old-deployment-uid" + status := newStatus(oldDep) + dep := newTestDeployment("openfga/openfga:v1.14.0") + r := newReconciler(t, nil, dep, status) + + reconcileOnce(t, r) + if _, err := getJob(r); err != nil { + t.Fatalf("expected a new Deployment to rerun the migration: %v", err) + } +} + +func TestReconcile_UnownedStatusCollision_ReturnsError(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + status := &corev1.ConfigMap{ + ObjectMeta: metav1.ObjectMeta{Name: statusKey.Name, Namespace: statusKey.Namespace}, + Data: map[string]string{"version": "v1.14.0"}, + } + r := newReconciler(t, nil, dep, status) + + if _, err := r.Reconcile(context.Background(), ctrl.Request{NamespacedName: deploymentKey}); err == nil { + t.Fatal("expected an unowned status ConfigMap collision to return an error") + } + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Errorf("the collision must prevent migration Job creation, got err=%v", err) + } +} + +func TestReconcile_StartedJobWithOutdatedTemplate_Kept(t *testing.T) { + for _, tt := range []struct { + name string + status batchv1.JobStatus + }{ + {"pod running", batchv1.JobStatus{Active: 1, Ready: ptr.To(int32(1))}}, + {"pod finished before the job is marked complete", batchv1.JobStatus{Succeeded: 1, Ready: ptr.To(int32(0))}}, + } { + t.Run(tt.name, func(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + job := newTestJob(dep) + job.Status = tt.status + dep.Spec.Template.Spec.Containers[0].Env[1].Value = "postgres://db.example.com/openfga" + r := newReconciler(t, nil, dep, job) + + if result := reconcileOnce(t, r); result.RequeueAfter != 10*time.Second { + t.Errorf("expected the job to be polled, got %v", result.RequeueAfter) + } + if _, err := getJob(r); err != nil { + t.Errorf("a started migration must not be replaced: %v", err) + } + }) + } +} + +func TestReconcile_RunningJobForPreviousVersion_KeptUntilItEnds(t *testing.T) { + // The image changes from v1.14.0 to v1.15.0 while the v1.14.0 migration is + // running. Interrupting it could leave the schema half-migrated, so the Job + // must be left alone and only replaced once it finishes. + oldDep := newTestDeployment("openfga/openfga:v1.14.0") + job := newTestJob(oldDep) + job.Status.Active = 1 + job.Status.Ready = ptr.To(int32(1)) + dep := newTestDeployment("openfga/openfga:v1.15.0") + r := newReconciler(t, nil, dep, job) + + if result := reconcileOnce(t, r); result.RequeueAfter != 10*time.Second { + t.Errorf("expected the running job to be polled, got %v", result.RequeueAfter) + } + kept, err := getJob(r) + if err != nil { + t.Fatalf("a running migration must not be deleted on a version change: %v", err) + } + if _, err := getStatus(r); !apierrors.IsNotFound(err) { + t.Errorf("no version may be recorded while the old job runs, got err=%v", err) + } + + kept.Status = batchv1.JobStatus{Succeeded: 1, Ready: ptr.To(int32(0)), Conditions: []batchv1.JobCondition{jobCondition(batchv1.JobComplete, time.Now())}} + if err := r.Status().Update(context.Background(), kept); err != nil { + t.Fatal(err) + } + reconcileOnce(t, r) + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Fatalf("expected the finished v1.14.0 job to be replaced, got err=%v", err) + } + if _, err := getStatus(r); !apierrors.IsNotFound(err) { + t.Errorf("a v1.14.0 job must not be recorded as a v1.15.0 migration, got err=%v", err) + } + + reconcileOnce(t, r) + replacement, err := getJob(r) + if err != nil { + t.Fatalf("expected a migration job for the new version: %v", err) + } + if got := replacement.Annotations[AnnotationDesiredVersion]; got != "v1.15.0" { + t.Errorf("expected the new job to target v1.15.0, got %q", got) + } +} + +func TestReconcile_UnownedJobCollision_ReturnsError(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + external := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Name: jobKey.Name, Namespace: jobKey.Namespace}} + r := newReconciler(t, nil, dep, external) + + if _, err := r.Reconcile(context.Background(), ctrl.Request{NamespacedName: deploymentKey}); err == nil { + t.Fatal("expected an unowned Job collision to return an error") + } + if _, err := getJob(r); err != nil { + t.Errorf("the unowned Job must not be deleted: %v", err) + } +} + +func TestReconcile_JobOwnedByPreviousDeployment_ReplacedWithPreconditions(t *testing.T) { + oldDep := newTestDeployment("openfga/openfga:v1.14.0") + oldDep.UID = "old-deployment-uid" + job := newTestJob(oldDep) + job.UID = "old-job-uid" + job.ResourceVersion = "7" + dep := newTestDeployment("openfga/openfga:v1.14.0") + var checkedPreconditions bool + r := newReconciler(t, &interceptor.Funcs{ + Delete: func(ctx context.Context, c client.WithWatch, obj client.Object, opts ...client.DeleteOption) error { + applied := (&client.DeleteOptions{}).ApplyOptions(opts) + if applied.Preconditions == nil || + applied.Preconditions.UID == nil || + *applied.Preconditions.UID != job.UID || + applied.Preconditions.ResourceVersion == nil || + *applied.Preconditions.ResourceVersion != job.ResourceVersion { + return fmt.Errorf("missing delete preconditions: %+v", applied.Preconditions) + } + if applied.PropagationPolicy == nil || *applied.PropagationPolicy != metav1.DeletePropagationForeground { + return fmt.Errorf("expected foreground deletion, got %v", applied.PropagationPolicy) + } + checkedPreconditions = true + return c.Delete(ctx, obj, opts...) + }, + }, dep, job) + + reconcileOnce(t, r) + if !checkedPreconditions { + t.Fatal("expected Job deletion to include UID and resourceVersion preconditions") + } + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Errorf("expected the stale operator Job to be deleted, got err=%v", err) + } +} + +func TestReconcile_TriggerChange_RunsMigrationAgain(t *testing.T) { + // Same image, but the chart's datastore configuration changed (or the user + // bumped migration.trigger): the recorded identity no longer matches. + dep := newTestDeployment("openfga/openfga:v1.14.0") + status := newStatus(dep) + dep.Annotations[AnnotationMigrationTrigger] = "b" + r := newReconciler(t, nil, dep, status) + + reconcileOnce(t, r) + job, err := getJob(r) + if err != nil { + t.Fatalf("expected a migration job for the new trigger: %v", err) + } + if job.Annotations[AnnotationMigrationTrigger] != "b" { + t.Errorf("expected the job to carry trigger b, got %q", job.Annotations[AnnotationMigrationTrigger]) + } + + job.Status = batchv1.JobStatus{Succeeded: 1, Conditions: []batchv1.JobCondition{jobCondition(batchv1.JobComplete, time.Now())}} + if err := r.Status().Update(context.Background(), job); err != nil { + t.Fatal(err) + } + reconcileOnce(t, r) + cm, _ := getStatus(r) + if cm.Data["version"] != "v1.14.0" || cm.Data["trigger"] != "b" { + t.Errorf("expected version v1.14.0 and trigger b recorded, got %v", cm.Data) + } +} + +func TestReconcile_UnownedJob_NotTouched(t *testing.T) { + foreign := &batchv1.Job{ + ObjectMeta: metav1.ObjectMeta{Name: jobKey.Name, Namespace: jobKey.Namespace, Labels: map[string]string{"app": "someone-else"}}, + Status: batchv1.JobStatus{Conditions: []batchv1.JobCondition{jobCondition(batchv1.JobComplete, time.Now())}}, + } + r := newReconciler(t, nil, newTestDeployment("openfga/openfga:v1.14.0"), foreign) + + if _, err := r.Reconcile(context.Background(), ctrl.Request{NamespacedName: deploymentKey}); err == nil { + t.Fatal("expected an error for a job the operator does not manage") + } + if _, err := getJob(r); err != nil { + t.Errorf("a job owned by someone else must not be deleted: %v", err) + } + if _, err := getStatus(r); !apierrors.IsNotFound(err) { + t.Errorf("a foreign job must not be recorded as a migration, got err=%v", err) + } + if evs := events(r); len(evs) != 1 || !strings.Contains(evs[0], "MigrationJobConflict") { + t.Errorf("expected a MigrationJobConflict event, got %v", evs) + } +} + +func TestBuildMigrationJob_SidecarsAndMetadata(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + dep.Spec.Template.Spec.InitContainers = []corev1.Container{{Name: "wait-for-db", Image: "busybox"}} + dep.Spec.Template.Spec.Containers = append(dep.Spec.Template.Spec.Containers, + corev1.Container{Name: "cloud-sql-proxy", Image: "gcr.io/cloud-sql-connectors/cloud-sql-proxy:2"}) + dep.Annotations[AnnotationMigrationLabels] = `{"team":"auth","app.kubernetes.io/component":"hijack"}` + dep.Annotations[AnnotationMigrationAnnotations] = `{"sidecar.istio.io/inject":"false","openfga.dev/desired-version":"spoof"}` + + job := newTestJob(dep) + inits := job.Spec.Template.Spec.InitContainers + if len(inits) != 2 || inits[0].Name != "cloud-sql-proxy" || inits[1].Name != "wait-for-db" { + t.Fatalf("expected the proxy sidecar then the init container, got %+v", inits) + } + // A native sidecar is stopped when the migrate container exits, so the Job completes. + if ptr.Deref(inits[0].RestartPolicy, "") != corev1.ContainerRestartPolicyAlways { + t.Errorf("expected the sidecar to have restartPolicy Always, got %v", inits[0].RestartPolicy) + } + if inits[1].RestartPolicy != nil { + t.Errorf("init containers keep their restart policy, got %v", *inits[1].RestartPolicy) + } + if len(job.Spec.Template.Spec.Containers) != 1 || job.Spec.Template.Spec.Containers[0].Name != "migrate-database" { + t.Errorf("expected only the migrate container, got %+v", job.Spec.Template.Spec.Containers) + } + for _, meta := range []metav1.ObjectMeta{job.ObjectMeta, job.Spec.Template.ObjectMeta} { + if meta.Labels["team"] != "auth" || meta.Annotations["sidecar.istio.io/inject"] != "false" { + t.Errorf("expected user metadata to be forwarded, got labels=%v annotations=%v", meta.Labels, meta.Annotations) + } + if meta.Labels[LabelComponent] != "migration" { + t.Errorf("operator identity labels must win, got %v", meta.Labels) + } + } + if job.Annotations[AnnotationDesiredVersion] != "v1.14.0" { + t.Errorf("operator annotations must win, got %v", job.Annotations) + } +} + +func TestReconcile_InvalidMetadataAnnotation_ReturnsError(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + dep.Annotations[AnnotationMigrationLabels] = "not json" + r := newReconciler(t, nil, dep) + if _, err := r.Reconcile(context.Background(), ctrl.Request{NamespacedName: deploymentKey}); err == nil { + t.Fatal("expected an error for an unparsable migration-labels annotation") + } +} + +// The legacy chart's Helm hook Job has the same name and carries the chart's +// app.kubernetes.io/version label, which is the chart appVersion rather than +// the image it ran. It must never be taken as proof of a migration. +func TestReconcile_LegacyHookJob_NotTrusted(t *testing.T) { + legacy := &batchv1.Job{ + ObjectMeta: metav1.ObjectMeta{ + Name: jobKey.Name, + Namespace: jobKey.Namespace, + Labels: map[string]string{ + "app.kubernetes.io/version": "v1.14.0", + "app.kubernetes.io/managed-by": "Helm", + }, + Annotations: map[string]string{"helm.sh/hook": "post-install, post-upgrade, post-rollback, post-delete"}, + }, + Status: batchv1.JobStatus{Conditions: []batchv1.JobCondition{jobCondition(batchv1.JobComplete, time.Now())}}, + } + r := newReconciler(t, nil, newTestDeployment("openfga/openfga:v1.14.0"), legacy) + + reconcileOnce(t, r) + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Errorf("expected the legacy hook job to be deleted, got err=%v", err) + } + if _, err := getStatus(r); !apierrors.IsNotFound(err) { + t.Errorf("the legacy hook job must not be recorded as a migration, got err=%v", err) + } + + reconcileOnce(t, r) + job, err := getJob(r) + if err != nil { + t.Fatalf("expected a new migration job: %v", err) + } + if job.Annotations[AnnotationDesiredVersion] != "v1.14.0" { + t.Errorf("expected the new job to target v1.14.0, got %v", job.Annotations) + } +} + +func TestReconcile_MigrationNotEnabled_Skips(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + delete(dep.Annotations, AnnotationMigrationEnabled) + r := newReconciler(t, nil, dep) + + if result := reconcileOnce(t, r); result.RequeueAfter != 0 { + t.Errorf("expected no requeue, got %v", result.RequeueAfter) + } + if _, err := getJob(r); !apierrors.IsNotFound(err) { + t.Errorf("expected no migration job, got err=%v", err) + } +} + +func TestReconcile_DeploymentNotFound_NoError(t *testing.T) { + r := newReconciler(t, nil) + if result := reconcileOnce(t, r); result.RequeueAfter != 0 { + t.Errorf("expected no requeue, got %v", result.RequeueAfter) + } +} + +func TestReconcile_ContainerFromAnnotation(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + dep.Annotations[AnnotationContainerName] = "server" + dep.Spec.Template.Spec.Containers = []corev1.Container{ + {Name: "sidecar", Image: "envoyproxy/envoy:v1.30.0"}, + {Name: "server", Image: "openfga/openfga:v1.14.0"}, + } + r := newReconciler(t, nil, dep) + reconcileOnce(t, r) + + job, err := getJob(r) + if err != nil { + t.Fatalf("expected migration job to be created: %v", err) + } + if image := job.Spec.Template.Spec.Containers[0].Image; image != "openfga/openfga:v1.14.0" { + t.Errorf("expected the annotated container's image, got %s", image) + } +} + +func TestReconcile_ContainerNotFound_ReturnsError(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + dep.Spec.Template.Spec.Containers[0].Name = "server" + r := newReconciler(t, nil, dep) + + if _, err := r.Reconcile(context.Background(), ctrl.Request{NamespacedName: deploymentKey}); err == nil { + t.Fatal("expected an error when the OpenFGA container is missing") + } +} + +func TestBuildMigrationJob_UsesMigrationPodConfiguration(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + dep.Spec.Template.Spec.Volumes = []corev1.Volume{{ + Name: "shared", + VolumeSource: corev1.VolumeSource{ + EmptyDir: &corev1.EmptyDirVolumeSource{}, + }, + }} + dep.Spec.Template.Spec.Containers[0].VolumeMounts = []corev1.VolumeMount{{Name: "shared", MountPath: "/shared"}} + dep.Spec.Template.Spec.Containers = append(dep.Spec.Template.Spec.Containers, corev1.Container{Name: "database-proxy", Image: "proxy:v1"}) + dep.Annotations[AnnotationMigrationInitContainers] = `[{"name":"prepare-proxy","image":"busybox:1.36"}]` + dep.Annotations[AnnotationMigrationSidecars] = `[{"name":"database-proxy","image":"proxy:v2"}]` + dep.Annotations[AnnotationMigrationVolumes] = `[{"name":"credentials","secret":{"secretName":"database-proxy"}}]` + dep.Annotations[AnnotationMigrationVolumeMounts] = `[{"name":"credentials","mountPath":"/credentials","readOnly":true}]` + dep.Annotations[AnnotationMigrationResources] = `{"requests":{"cpu":"100m"}}` + dep.Annotations[AnnotationMigrationTimeout] = "2m" + dep.Annotations[AnnotationMigrationTrigger] = "secret-rotation-2" + dep.Annotations[AnnotationMigrationAnnotations] = `{"admission.example.com/inject":"enabled","openfga.dev/migration-trigger":"ignored"}` + dep.Annotations[AnnotationMigrationLabels] = `{"app.kubernetes.io/component":"overridden","network-policy.example.com/database":"allowed"}` + + job := newTestJob(dep) + spec := job.Spec.Template.Spec + if len(spec.InitContainers) != 2 || spec.InitContainers[0].Name != "database-proxy" || spec.InitContainers[1].Name != "prepare-proxy" { + t.Errorf("unexpected init containers: %+v", spec.InitContainers) + } + if ptr.Deref(spec.InitContainers[0].RestartPolicy, "") != corev1.ContainerRestartPolicyAlways { + t.Errorf("expected migration sidecar to use native sidecar semantics, got %v", spec.InitContainers[0].RestartPolicy) + } + if spec.InitContainers[0].Image != "proxy:v2" { + t.Errorf("expected migration sidecar to override the runtime sidecar, got %q", spec.InitContainers[0].Image) + } + if len(spec.Containers) != 1 || spec.Containers[0].Name != "migrate-database" { + t.Errorf("unexpected containers: %+v", spec.Containers) + } + if len(spec.Volumes) != 2 || spec.Volumes[1].Name != "credentials" { + t.Errorf("unexpected volumes: %+v", spec.Volumes) + } + migrate := spec.Containers[0] + if len(migrate.VolumeMounts) != 2 || migrate.VolumeMounts[1].MountPath != "/credentials" { + t.Errorf("unexpected migration volume mounts: %+v", migrate.VolumeMounts) + } + if got := migrate.Resources.Requests[corev1.ResourceCPU]; got.Cmp(resource.MustParse("100m")) != 0 { + t.Errorf("expected 100m CPU request, got %s", got.String()) + } + if !hasEnvVar(migrate.Env, "OPENFGA_TIMEOUT") { + t.Errorf("expected OPENFGA_TIMEOUT in %+v", migrate.Env) + } + if got := job.Spec.Template.Annotations[AnnotationMigrationTrigger]; got != "secret-rotation-2" { + t.Errorf("expected migration trigger on the Job pod template, got %q", got) + } + if got := job.Annotations["admission.example.com/inject"]; got != "enabled" { + t.Errorf("expected custom Job annotation, got %q", got) + } + if got := job.Spec.Template.Annotations["admission.example.com/inject"]; got != "enabled" { + t.Errorf("expected custom pod annotation, got %q", got) + } + if got := job.Labels[LabelComponent]; got != "migration" { + t.Errorf("operator Job label must take precedence, got %q", got) + } + if got := job.Spec.Template.Labels[LabelComponent]; got != "migration" { + t.Errorf("operator pod label must take precedence, got %q", got) + } + if got := job.Spec.Template.Labels["network-policy.example.com/database"]; got != "allowed" { + t.Errorf("expected custom pod label, got %q", got) + } +} + +func TestReconcile_InvalidMigrationPodConfiguration_ReturnsError(t *testing.T) { + for _, value := range []string{`not-json`, `[{"name":"proxy","image":"proxy:v2","restartPolcy":"Always"}]`, `[] {}`} { + t.Run(value, func(t *testing.T) { + dep := newTestDeployment("openfga/openfga:v1.14.0") + dep.Annotations[AnnotationMigrationSidecars] = value + r := newReconciler(t, nil, dep) + + if _, err := r.Reconcile(context.Background(), ctrl.Request{NamespacedName: deploymentKey}); err == nil { + t.Fatal("expected invalid migration sidecars to return an error") + } + }) + } +} + +func TestExtractImageTag(t *testing.T) { + tests := []struct { + image string + expected string + }{ + {"openfga/openfga:v1.14.0", "v1.14.0"}, + {"openfga/openfga:latest", "latest"}, + {"openfga/openfga", "latest"}, + {"openfga", "latest"}, + {"ghcr.io/openfga/openfga:v1.14.0", "v1.14.0"}, + {"registry.example.com:5000/openfga/openfga:v1.14.0", "v1.14.0"}, + {"registry.example.com:5000/openfga/openfga", "latest"}, + {"openfga/openfga@sha256:abcdef1234567890", "sha256:abcdef1234567890"}, + {"openfga/openfga:v1.14.0@sha256:abcdef1234567890", "sha256:abcdef1234567890"}, + } + for _, tt := range tests { + t.Run(tt.image, func(t *testing.T) { + if got := extractImageTag(tt.image); got != tt.expected { + t.Errorf("extractImageTag(%q) = %q, want %q", tt.image, got, tt.expected) + } + }) + } +} + +func TestClearMigrationFailedConditionIdempotent(t *testing.T) { + dep := &appsv1.Deployment{} + if clearMigrationFailedCondition(dep) { + t.Error("expected no change when the MigrationFailed condition is absent") + } + if len(dep.Status.Conditions) != 0 { + t.Errorf("expected no conditions to be added, got %d", len(dep.Status.Conditions)) + } + + dep.Status.Conditions = []appsv1.DeploymentCondition{{Type: "MigrationFailed", Status: corev1.ConditionTrue}} + if !clearMigrationFailedCondition(dep) { + t.Error("expected a change when clearing a True MigrationFailed condition") + } + cond := findCondition(dep.Status.Conditions, "MigrationFailed") + if cond == nil || cond.Status != corev1.ConditionFalse { + t.Fatalf("expected MigrationFailed=False after clear, got %+v", cond) + } + transition := cond.LastTransitionTime + + if clearMigrationFailedCondition(dep) { + t.Error("expected no change when the MigrationFailed condition is already False") + } + if cond := findCondition(dep.Status.Conditions, "MigrationFailed"); !cond.LastTransitionTime.Equal(&transition) { + t.Error("LastTransitionTime must not change when the condition is already False") + } +} + +func TestSetMigrationFailedConditionIdempotent(t *testing.T) { + dep := &appsv1.Deployment{} + if !setMigrationFailedCondition(dep, "v1.14.0") { + t.Error("expected a change when setting MigrationFailed on a fresh deployment") + } + cond := findCondition(dep.Status.Conditions, "MigrationFailed") + if cond == nil || cond.Status != corev1.ConditionTrue { + t.Fatalf("expected MigrationFailed=True, got %+v", cond) + } + transition := cond.LastTransitionTime + + if setMigrationFailedCondition(dep, "v1.14.0") { + t.Error("expected no change when re-setting the same MigrationFailed condition") + } + if !setMigrationFailedCondition(dep, "v1.15.0") { + t.Error("expected a change when the failure message changes") + } + if cond := findCondition(dep.Status.Conditions, "MigrationFailed"); !cond.LastTransitionTime.Equal(&transition) { + t.Error("LastTransitionTime must not change without a status transition") + } +} diff --git a/operator/tests/README.md b/operator/tests/README.md new file mode 100644 index 00000000..9a225f7f --- /dev/null +++ b/operator/tests/README.md @@ -0,0 +1,185 @@ +# Local Integration Tests + +Manual integration tests for the OpenFGA operator on a local Kubernetes cluster (Rancher Desktop, kind, minikube, etc.). + +## Prerequisites + +- A running local Kubernetes cluster +- Helm 3.6+ +- The operator image built locally: + ```bash + cd operator + docker build -t openfga/openfga-operator:dev . + ``` +- Chart dependencies updated: + ```bash + helm dependency update charts/openfga + ``` + +All test values files use `imagePullPolicy: Never`, so the locally-built image must be available to the cluster's container runtime. On Rancher Desktop (dockerd) and Docker Desktop this works automatically. For kind, load the image first: + +```bash +kind load docker-image openfga/openfga-operator:dev +``` + +## Test Scenarios + +### 1. Happy Path + +Deploys OpenFGA with a Postgres instance. The operator should run the migration and all OpenFGA pods should become ready within ~30 seconds. + +```bash +kubectl create namespace openfga-test +helm install openfga-test charts/openfga -n openfga-test \ + -f operator/tests/values-happy-path.yaml +``` + +**Expected outcome:** + +| Resource | State | +|----------|-------| +| `openfga-test-openfga-operator` | `1/1 Running` | +| `openfga-test-postgres` | `1/1 Running` | +| `openfga-test-migrate` | `0/1 Completed` | +| `openfga-test` (OpenFGA) | `3/3 Running` | + +**Verify:** + +```bash +# All resources healthy +kubectl get all -n openfga-test + +# Operator logs show full lifecycle +kubectl logs -n openfga-test deployment/openfga-test-openfga-operator + +# Migration status recorded +kubectl get configmap openfga-test-migration-status -n openfga-test -o jsonpath='{.data}' + +# Database tables created +kubectl exec -n openfga-test deployment/openfga-test-postgres -- \ + psql -U openfga -d openfga -c '\dt' + +# OpenFGA responding +kubectl run curl-test --image=curlimages/curl -n openfga-test \ + --rm -it --restart=Never -- curl -s http://openfga-test:8080/healthz +# Expected: {"status":"SERVING"} +``` + +**Clean up:** + +```bash +helm uninstall openfga-test -n openfga-test +kubectl delete namespace openfga-test +``` + +--- + +### 2. Database Outage and Recovery + +Deploys OpenFGA with a Postgres instance scaled to 0 replicas (simulating a database that isn't ready yet). The operator should retry migrations until Postgres becomes available, then self-heal. + +```bash +kubectl create namespace openfga-test +helm install openfga-test charts/openfga -n openfga-test \ + -f operator/tests/values-db-outage.yaml +``` + +**Expected behavior while Postgres is down:** + +- Migration Job runs and fails (each pod times out after ~60s) +- After 3 failures (backoffLimit), the operator: + - Sets `MigrationFailed: True` condition on the Deployment + - Keeps the failed Job for 60 seconds, then replaces it with a fresh one +- This cycle repeats indefinitely +- OpenFGA stays at 0/3 throughout: the pods cannot reach the database, so they restart and never pass the readiness check + +**Watch the failure cycle:** + +```bash +# Check deployment conditions +kubectl get deployment openfga-test -n openfga-test \ + -o jsonpath='{range .status.conditions[*]}{.type}: {.status} - {.message}{"\n"}{end}' + +# Watch operator logs for delete/retry cycle +kubectl logs -n openfga-test deployment/openfga-test-openfga-operator -f +# Look for: +# "migration job failed" +# "retrying migration" +# "created migration job" +``` + +**Bring Postgres back (after a few minutes):** + +```bash +kubectl scale deployment openfga-test-postgres -n openfga-test --replicas=1 +``` + +**Expected recovery:** + +- The next migration Job connects and succeeds (within ~60s of Postgres becoming ready) +- Operator updates the ConfigMap with the new version and sets `MigrationFailed: False` +- OpenFGA pods become ready once their restart back-off expires (up to 5 minutes) +- `{"status":"SERVING"}` from the health endpoint + +**Verify recovery:** + +```bash +# OpenFGA should be 3/3 Running +kubectl get all -n openfga-test + +# Migration status recorded +kubectl get configmap openfga-test-migration-status -n openfga-test -o jsonpath='{.data}' + +# Health check +kubectl run curl-test --image=curlimages/curl -n openfga-test \ + --rm -it --restart=Never -- curl -s http://openfga-test:8080/healthz +``` + +**Clean up:** + +```bash +helm uninstall openfga-test -n openfga-test +kubectl delete namespace openfga-test +``` + +--- + +### 3. No Database (Permanent Failure) + +Deploys OpenFGA pointing at a Postgres hostname that doesn't exist. The operator should continuously retry without crashing or leaving the app in a broken state. + +```bash +kubectl create namespace openfga-test +helm install openfga-test charts/openfga -n openfga-test \ + -f operator/tests/values-no-db.yaml +``` + +**Expected behavior:** + +- Migration Jobs fail repeatedly (DNS resolution fails for `postgres-does-not-exist`) +- Operator sets `MigrationFailed: True` on the Deployment +- Operator replaces each failed Job 60 seconds after it fails +- OpenFGA stays at 0/3 indefinitely and never serves traffic + +This scenario verifies the operator doesn't crash-loop or consume excessive resources when the database is permanently unavailable. + +**Verify:** + +```bash +# OpenFGA at 0/3 (pods NotReady), operator at 1/1 +kubectl get deployments -n openfga-test + +# MigrationFailed condition present +kubectl get deployment openfga-test -n openfga-test \ + -o jsonpath='{range .status.conditions[*]}{.type}: {.status} - {.message}{"\n"}{end}' + +# Operator logs show retry cycle +kubectl logs -n openfga-test deployment/openfga-test-openfga-operator --tail=20 +``` + +**Clean up:** + +```bash +helm uninstall openfga-test -n openfga-test +kubectl delete namespace openfga-test +``` diff --git a/operator/tests/values-db-outage.yaml b/operator/tests/values-db-outage.yaml new file mode 100644 index 00000000..16d7427e --- /dev/null +++ b/operator/tests/values-db-outage.yaml @@ -0,0 +1,63 @@ +# Test values: Postgres deployed but scaled to 0 (simulates DB outage) +openfga-operator: + enabled: true + image: + repository: openfga/openfga-operator + tag: dev + pullPolicy: Never + resources: + requests: + cpu: 10m + memory: 64Mi + +datastore: + engine: postgres + uri: "postgres://openfga:changeme@openfga-test-postgres:5432/openfga?sslmode=disable" + +extraObjects: + - apiVersion: v1 + kind: Secret + metadata: + name: openfga-test-postgres-creds + stringData: + POSTGRES_USER: openfga + POSTGRES_PASSWORD: changeme + POSTGRES_DB: openfga + - apiVersion: apps/v1 + kind: Deployment + metadata: + name: openfga-test-postgres + spec: + replicas: 0 # Start with Postgres DOWN + selector: + matchLabels: + app: openfga-test-postgres + template: + metadata: + labels: + app: openfga-test-postgres + spec: + containers: + - name: postgres + image: postgres:17 + ports: + - containerPort: 5432 + envFrom: + - secretRef: + name: openfga-test-postgres-creds + volumeMounts: + - name: data + mountPath: /var/lib/postgresql/data + volumes: + - name: data + emptyDir: {} + - apiVersion: v1 + kind: Service + metadata: + name: openfga-test-postgres + spec: + selector: + app: openfga-test-postgres + ports: + - port: 5432 + targetPort: 5432 diff --git a/operator/tests/values-happy-path.yaml b/operator/tests/values-happy-path.yaml new file mode 100644 index 00000000..53f32717 --- /dev/null +++ b/operator/tests/values-happy-path.yaml @@ -0,0 +1,63 @@ +# Local test values for operator-managed migration on Rancher Desktop +openfga-operator: + enabled: true + image: + repository: openfga/openfga-operator + tag: dev + pullPolicy: Never + resources: + requests: + cpu: 10m + memory: 64Mi + +datastore: + engine: postgres + uri: "postgres://openfga:changeme@openfga-test-postgres:5432/openfga?sslmode=disable" + +extraObjects: + - apiVersion: v1 + kind: Secret + metadata: + name: openfga-test-postgres-creds + stringData: + POSTGRES_USER: openfga + POSTGRES_PASSWORD: changeme + POSTGRES_DB: openfga + - apiVersion: apps/v1 + kind: Deployment + metadata: + name: openfga-test-postgres + spec: + replicas: 1 + selector: + matchLabels: + app: openfga-test-postgres + template: + metadata: + labels: + app: openfga-test-postgres + spec: + containers: + - name: postgres + image: postgres:17 + ports: + - containerPort: 5432 + envFrom: + - secretRef: + name: openfga-test-postgres-creds + volumeMounts: + - name: data + mountPath: /var/lib/postgresql/data + volumes: + - name: data + emptyDir: {} + - apiVersion: v1 + kind: Service + metadata: + name: openfga-test-postgres + spec: + selector: + app: openfga-test-postgres + ports: + - port: 5432 + targetPort: 5432 diff --git a/operator/tests/values-no-db.yaml b/operator/tests/values-no-db.yaml new file mode 100644 index 00000000..e43000cd --- /dev/null +++ b/operator/tests/values-no-db.yaml @@ -0,0 +1,16 @@ +# Test values with NO postgres — simulates database unavailable +openfga-operator: + enabled: true + image: + repository: openfga/openfga-operator + tag: dev + pullPolicy: Never + resources: + requests: + cpu: 10m + memory: 64Mi + +datastore: + engine: postgres + # Points to a service that doesn't exist + uri: "postgres://openfga:changeme@postgres-does-not-exist:5432/openfga?sslmode=disable"