Compare commits
46 Commits
076bc66ebe
...
feat/demo-
| Author | SHA1 | Date | |
|---|---|---|---|
| f6b67006dc | |||
| 0d68111bc2 | |||
| e1abac3e8f | |||
| f3ff6a0ca2 | |||
| 09845e4eaf | |||
| e7fdae0859 | |||
| d2c317344e | |||
| 9230b1213c | |||
| 57e3ea22cf | |||
| 95c487415b | |||
| 849ec1083e | |||
| f7000f7514 | |||
| 19d6a8dfba | |||
| 420c3509b0 | |||
| ed59a4c384 | |||
| 4619c352c0 | |||
| 5a7f0a30c3 | |||
| ae434a7167 | |||
| e4d2a191d0 | |||
| 837e374228 | |||
| c137028364 | |||
| c108a06a94 | |||
| 0fe62ef314 | |||
| d595a93d36 | |||
| c489832ce7 | |||
| add120c033 | |||
| 8176a5eef8 | |||
| 4aa3d47e3c | |||
| f6d50e4744 | |||
| 223b6a8fd6 | |||
| 801a9fbe5f | |||
| 71c00c40d1 | |||
| 1125f74221 | |||
| 5c408cc284 | |||
| 4282d73c74 | |||
| 05d490c0b8 | |||
| 7700358764 | |||
| ff859ebd84 | |||
| 4529594fb7 | |||
| ef1387dc01 | |||
| a4a483acbc | |||
| f28766fce3 | |||
| 26991fbe13 | |||
| 652f155baf | |||
| e614264597 | |||
| a5f0aa95d2 |
15
.claude/agents/operator-reviewer.md
Normal file
15
.claude/agents/operator-reviewer.md
Normal file
@@ -0,0 +1,15 @@
|
||||
---
|
||||
name: operator-reviewer
|
||||
description: Reviews Kubernetes operator PRs for controller-runtime correctness, reconcile semantics, and API design
|
||||
tools: Read, Grep, Glob, Bash
|
||||
---
|
||||
You are a senior reviewer specializing in Kubernetes operators.
|
||||
Review with focus on:
|
||||
- Reconcile idempotency and requeue behavior; no state assumptions between reconciles
|
||||
- Informer cache reads vs direct API reads; stale-cache races
|
||||
- Finalizer handling, deletion flow, orphaned resources
|
||||
- CRD schema evolution, conversion webhooks, status subresource / conditions conventions
|
||||
- RBAC minimality vs what the controller actually touches
|
||||
- Leader election, watch predicates, event filtering for churn reduction
|
||||
- Go: context propagation, error wrapping, client.Object handling
|
||||
Output: findings ranked by severity, with file:line refs. No praise padding.
|
||||
@@ -39,11 +39,59 @@
|
||||
"Bash(mkdir -p docs/plans && date \"+%Y-%m-%d-%H%M\")",
|
||||
"Bash(go install *)",
|
||||
"Bash(go env *)",
|
||||
"Bash(git rm *)"
|
||||
"Bash(git rm *)",
|
||||
"Bash(mkdir -p /Users/jan.novak/srv/go/egress-proxies-operator/docs/plans-executions)",
|
||||
"Bash(cd /Users/jan.novak/srv/go/egress-proxies-operator *)",
|
||||
"Bash(echo \"build: $?\")",
|
||||
"Bash(echo \"vet: $?\")",
|
||||
"Bash(perl -i -pe 's{^\\\\t\\\\t\\\\t\\\\t\\\\t// TODO\\\\\\(user\\\\\\): Specify other spec details if needed\\\\.\\\\n}{\\\\t\\\\t\\\\t\\\\t\\\\t// A minimal, schema-valid spec so this placeholder test survives the\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t// CEL/CRD validation added in Step 1. Rewritten wholesale in Step 4\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t// alongside the real reconciler and envtest suite.\\\\n\\\\t\\\\t\\\\t\\\\t\\\\tSpec: crawlv1alpha1.ProxySpec{\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t\\\\tMode: crawlv1alpha1.ModeExternal,\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t\\\\tEndpoint: &crawlv1alpha1.EndpointSpec{Host: \"10.0.0.1\"},\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t},\\\\n}' internal/controller/proxy_controller_test.go)",
|
||||
"Bash(go tool *)",
|
||||
"Bash(docker version *)",
|
||||
"Bash(curl -sI --max-time 5 https://hub.docker.com)",
|
||||
"Bash(kind get *)",
|
||||
"Bash(curl -s --max-time 5 \"https://hub.docker.com/v2/repositories/ubuntu/squid/tags?page_size=10\")",
|
||||
"Bash(python3 -c ' *)",
|
||||
"Bash(curl -s --max-time 5 \"https://hub.docker.com/v2/repositories/ubuntu/squid/tags?page_size=100\")",
|
||||
"Bash(go doc *)",
|
||||
"Bash(go list *)",
|
||||
"Bash(gofmt -w internal/provider/config.go)",
|
||||
"Bash(gofmt -l .)",
|
||||
"Bash(git restore *)",
|
||||
"Bash(make manifests *)",
|
||||
"Bash(make test *)",
|
||||
"Bash(KUBEBUILDER_ASSETS=\"/Users/jan.novak/srv/go/egress-proxies-operator/bin/k8s/1.36.2-darwin-arm64\" go test -race ./internal/controller/)",
|
||||
"Bash(grep -n 'func Channel' -A8 __CMDSUB_OUTPUT__/sigs.k8s.io/controller-runtime@v0.24.1/pkg/source/source.go)",
|
||||
"Bash(grep -n 'type GenericEvent' __CMDSUB_OUTPUT__/sigs.k8s.io/controller-runtime@v0.24.1/pkg/event/event.go)",
|
||||
"Bash(KUBEBUILDER_ASSETS=__TRACKED_VAR__/bin/k8s/1.36.2-darwin-arm64 go test -race ./...)",
|
||||
"Bash(cat >> *)",
|
||||
"Bash(make run-dev *)",
|
||||
"Bash(kubectl get *)",
|
||||
"Bash(kubectl delete *)",
|
||||
"Bash(make docker-build *)",
|
||||
"Bash(kind load *)",
|
||||
"Bash(make deploy *)",
|
||||
"Bash(kubectl -n egress-proxies-operator-system rollout status deploy/egress-proxies-operator-controller-manager --timeout=120s)",
|
||||
"Bash(kubectl -n egress-proxies-operator-system rollout restart deploy/egress-proxies-operator-controller-manager)",
|
||||
"Bash(git -C /Users/jan.novak/srv/go/egress-proxies-operator add docs/plans/2026-08-11-1742-gcp-provider-verbose-logging.md)",
|
||||
"Bash(git -C /Users/jan.novak/srv/go/egress-proxies-operator commit -m 'Add plan: verbose V-level logging in the GCP provider *)",
|
||||
"Bash(echo \"exit: $?\")",
|
||||
"Bash(echo \"tests exit: $?\")",
|
||||
"Bash(./bin/manager --version)",
|
||||
"Bash(./bin/manager-stamped --version)",
|
||||
"Bash(./bin/manager-pkg --version)",
|
||||
"Bash(docker run *)",
|
||||
"Bash(kubectl -n egress-proxies-operator-system get pods -o wide)",
|
||||
"Bash(kubectl -n egress-proxies-operator-system get deploy egress-proxies-operator-controller-manager -o jsonpath='{.spec.template.spec.containers[0].args}')",
|
||||
"Bash(kubectl -n egress-proxies-operator-system logs deploy/egress-proxies-operator-controller-manager)",
|
||||
"Bash(python3 -c \"import json; d=json.load\\(open\\('docs/deploy/sa_key.json'\\)\\); print\\(d.get\\('type'\\), d.get\\('client_email'\\)\\)\")"
|
||||
],
|
||||
"additionalDirectories": [
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/.claude",
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/docs/plans"
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/docs/plans",
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/docs",
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/docs/prompts",
|
||||
"/tmp",
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/docs/plans-executions"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
76
.gitea/workflows/build.yaml
Normal file
76
.gitea/workflows/build.yaml
Normal file
@@ -0,0 +1,76 @@
|
||||
name: Build and Push
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
tag:
|
||||
description: 'Image tag'
|
||||
required: true
|
||||
default: 'latest'
|
||||
push:
|
||||
tags:
|
||||
- '*'
|
||||
|
||||
concurrency:
|
||||
group: build-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
check:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
- name: Vet
|
||||
run: go vet ./...
|
||||
- name: Build
|
||||
run: go build ./...
|
||||
- name: Test (short)
|
||||
run: go test -short ./...
|
||||
|
||||
build:
|
||||
needs: check
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Compute image tags
|
||||
id: meta
|
||||
run: |
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
TAG="${{ inputs.tag }}"
|
||||
else
|
||||
TAG="${{ github.ref_name }}"
|
||||
fi
|
||||
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
|
||||
echo "sha=sha-$(echo '${{ github.sha }}' | cut -c1-12)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Login to Gitea registry
|
||||
run: echo "${{ secrets.REGISTRY_TOKEN }}" | docker login -u ${{ github.actor }} --password-stdin gitea.home.hrajfrisbee.cz
|
||||
|
||||
- name: Build and push
|
||||
run: |
|
||||
IMAGE=gitea.home.hrajfrisbee.cz/${{ github.repository }}
|
||||
docker build \
|
||||
--build-arg GIT_COMMIT=$(echo '${{ github.sha }}' | cut -c1-12) \
|
||||
--label org.opencontainers.image.source=https://gitea.home.hrajfrisbee.cz/${{ github.repository }} \
|
||||
--label org.opencontainers.image.created=$(date -u +%Y-%m-%dT%H:%M:%SZ) \
|
||||
-t "$IMAGE:${{ steps.meta.outputs.tag }}" \
|
||||
-t "$IMAGE:${{ steps.meta.outputs.sha }}" \
|
||||
.
|
||||
docker push "$IMAGE:${{ steps.meta.outputs.tag }}"
|
||||
docker push "$IMAGE:${{ steps.meta.outputs.sha }}"
|
||||
|
||||
# Only real tag pushes move :latest — an ad-hoc dispatch of an old ref must not clobber it.
|
||||
- name: Push latest (tag builds only)
|
||||
if: github.event_name == 'push'
|
||||
run: |
|
||||
IMAGE=gitea.home.hrajfrisbee.cz/${{ github.repository }}
|
||||
docker tag "$IMAGE:${{ steps.meta.outputs.tag }}" "$IMAGE:latest"
|
||||
docker push "$IMAGE:latest"
|
||||
3
.gitignore
vendored
3
.gitignore
vendored
@@ -28,3 +28,6 @@ go.work
|
||||
|
||||
# Kubeconfig might contain secrets
|
||||
*.kubeconfig
|
||||
|
||||
# GCP service-account keys (created per docs/gcp-in-specific-project.md)
|
||||
sa_key.json
|
||||
|
||||
24
CHANGELOG.md
24
CHANGELOG.md
@@ -1 +1,25 @@
|
||||
# Changelog
|
||||
|
||||
## 2026-08-10 09:34 CEST — kind e2e verified; fix Squid OOM in containers; quickstart goes in-cluster
|
||||
|
||||
- Full end-to-end pass on a throwaway kind cluster: Squid pod Ready with a real CONNECT
|
||||
probe (89 ms), lease grant/report/cooldown-409/release through the discovery API,
|
||||
finalizer cleanup on delete.
|
||||
- Fixed the kubernetes provider's generated squid.conf: `max_filedescriptors 1024`
|
||||
(squid sizes FD tables from the container's effectively-unlimited RLIMIT_NOFILE and
|
||||
was OOM-killed at startup under kind/containerd) + `cache_mem 16 MB`.
|
||||
- README quickstart now deploys the operator in-cluster: `make run-dev` on a laptop
|
||||
cannot reach kind pod IPs, so health probes fail by construction there (documented).
|
||||
|
||||
## 2026-08-09 17:27 CEST — Operator wired end to end: reconciler, health, leases, discovery, GC, two providers
|
||||
|
||||
- `cmd/main.go` is now the full composition root: `--providers-config` (required, fail-fast),
|
||||
`--discovery-addr`, `--proxy-namespace`, `--health-workers`, `--gc-interval`, `--gc-min-age`,
|
||||
`--gc-allow-namespaced`, `--lease-cooldown`, `--max-lease-ttl`; wires the kubernetes + gcp
|
||||
providers (metrics-instrumented), health engine, lease store, discovery API, orphan GC, and
|
||||
Prometheus metrics onto one manager.
|
||||
- Deploy manifests: providers ConfigMap mount, optional `DISCOVERY_TOKEN` Secret env,
|
||||
discovery port 8090 + Service; pods RBAC for the kubernetes provider.
|
||||
- Samples for all three proxy flavors + providers-config; `make run-dev` for local development.
|
||||
- README rewritten (kind quickstart, GCP setup, the two load-bearing caveats); architecture doc
|
||||
completed with components table and the full decision log.
|
||||
|
||||
71
CLAUDE.md
71
CLAUDE.md
@@ -86,6 +86,63 @@ Include the same timestamp in the plan's header, e.g.:
|
||||
Create `docs/plans/` on first use. Plan files are committed to the repo so other
|
||||
contributors can review historical decisions.
|
||||
|
||||
**Ordering matters:** plan mode itself restricts writes to the default
|
||||
`~/.claude/plans/` location — that's a hard tool restriction during plan mode, not a
|
||||
choice. So the copy into `docs/plans/` cannot happen until *after* `ExitPlanMode`.
|
||||
When it happens, it must be the very first action taken post-exit, committed on its
|
||||
own, *before* any implementation work starts (branching, scaffolding, editing code).
|
||||
Do not fold the plan-file commit into the first implementation commit, and do not
|
||||
defer it until "later" in the same turn — treat "save and commit the plan" as its own
|
||||
checkpoint that blocks starting Step 0.
|
||||
|
||||
### Plan execution summaries
|
||||
|
||||
After finishing each step of an approved plan, append a short, human-readable summary
|
||||
to `docs/plans-executions/YYYY-MM-DD-HHMM-<slug>.md` — same timestamp and slug as the
|
||||
originating plan in `docs/plans/`, so the two files pair up 1:1. Create the file (and
|
||||
`docs/plans-executions/` on first use) after the first step completes; append a new
|
||||
section per step after that, in chronological order (bottom of the file), not
|
||||
newest-first like `CHANGELOG.md`.
|
||||
|
||||
**Status checklist.** The top of the file, above the first step's entry, is a markdown
|
||||
checklist with one line per step from the plan (`- [ ] Step 0 — <title>`), so a single
|
||||
glance shows what's done and what's left without reading prose or cross-referencing
|
||||
`git log`. Check a box in the same commit that adds that step's entry — never leave a
|
||||
completed step unchecked or a not-yet-done step checked. This checklist is the
|
||||
authoritative status record; an in-session todo list (e.g. `TodoWrite`) is ephemeral
|
||||
and does not substitute for it — a new session resuming this plan should be able to
|
||||
read this file alone and know exactly where to pick up.
|
||||
|
||||
Keep the prose light — a few sentences on what was done, plus anything genuinely
|
||||
interesting or non-obvious about how it went (a judgment call made, a spec gap found,
|
||||
something that didn't work as expected). This is not a duplicate of the plan or the
|
||||
diff; skip steps that went exactly as planned with nothing worth flagging.
|
||||
|
||||
**Include exact commands.** Any invocation of an external tool that isn't a plain file
|
||||
edit — CLI scaffolding tools (`kubebuilder`, code generators), package installs, other
|
||||
non-obvious shell commands — goes in as a fenced code block with the exact
|
||||
command-line actually run, not a paraphrase. Prefer the real invocation over a
|
||||
description of it; a future reader (or a future Claude) should be able to copy the
|
||||
snippet and reproduce the step. Trim noisy stdout, but keep anything that changed the
|
||||
outcome (a flag that mattered, an unexpected error, a version that got picked
|
||||
automatically).
|
||||
|
||||
**Structure each command snippet with one line of context before it** — why it was run
|
||||
that way, not just that it was run (e.g. "installed into a scratch `GOBIN` because the
|
||||
plan's install path 404s" beats a bare command with no framing). When something
|
||||
deviated from what the plan said — a flag correction, a version resolved differently
|
||||
than expected, a tool auto-running another tool — call that out explicitly rather than
|
||||
presenting the final working command as if it were the first thing tried. Close the
|
||||
entry with a short "worth noting" paragraph covering anything a future reader should
|
||||
know before touching this step's output: defaults that came out differently than
|
||||
planned, files outside the plan's scope that got touched, things deliberately left for
|
||||
a later step.
|
||||
|
||||
Use the Step 0 entry in `docs/plans-executions/2026-08-07-1747-proxy-operator.md` as
|
||||
the reference example for the level of detail expected.
|
||||
|
||||
Commit the summary update together with that step's implementation commit.
|
||||
|
||||
### Changelog
|
||||
|
||||
Maintain a running changelog in `CHANGELOG.md` at the repo root. After every
|
||||
@@ -142,8 +199,18 @@ Always append a `Co-Authored-By` trailer to indicate AI assistance:
|
||||
|
||||
Co-Authored-By: Claude <noreply@anthropic.com>
|
||||
|
||||
TODO: no `.gitea/workflows/` CI pipeline exists yet — add a CI/CD subsection here once
|
||||
one is set up.
|
||||
### CI/CD
|
||||
|
||||
`.gitea/workflows/build.yaml` builds the manager image and pushes it to the Gitea
|
||||
registry. Triggers: any tag push, or manual `workflow_dispatch` with a `tag` input.
|
||||
A lightweight `check` job (`go vet` / `go build` / `go test -short`) gates the build.
|
||||
|
||||
- Images: `gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator:<tag>` plus an
|
||||
immutable `sha-<12-char-commit>` tag on every build; `:latest` moves only on real
|
||||
tag pushes, never on manual dispatch.
|
||||
- Requires the `REGISTRY_TOKEN` repo secret (Gitea PAT with `write:package`),
|
||||
same convention as the other projects on this Gitea instance.
|
||||
- The commit is baked into the binary via the `GIT_COMMIT` build arg (see Dockerfile).
|
||||
|
||||
## Gotchas
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
FROM golang:1.26 AS builder
|
||||
ARG TARGETOS
|
||||
ARG TARGETARCH
|
||||
ARG GIT_COMMIT=unknown
|
||||
|
||||
WORKDIR /workspace
|
||||
# Copy the Go Modules manifests
|
||||
@@ -19,11 +20,15 @@ COPY . .
|
||||
# was called. For example, if we call make docker-build in a local env which has the Apple Silicon M1 SO
|
||||
# the docker BUILDPLATFORM arg will be linux/arm64 when for Apple x86 it will be linux/amd64. Therefore,
|
||||
# by leaving it empty we can ensure that the container and binary shipped on it will have the same platform.
|
||||
RUN CGO_ENABLED=0 GOOS=${TARGETOS:-linux} GOARCH=${TARGETARCH} go build -a -o manager cmd/main.go
|
||||
RUN CGO_ENABLED=0 GOOS=${TARGETOS:-linux} GOARCH=${TARGETARCH} go build -a \
|
||||
-ldflags "-X gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/version.Commit=${GIT_COMMIT}" \
|
||||
-o manager cmd/main.go
|
||||
|
||||
# Use distroless as minimal base image to package the manager binary
|
||||
# Refer to https://github.com/GoogleContainerTools/distroless for more details
|
||||
FROM gcr.io/distroless/static:nonroot
|
||||
ARG GIT_COMMIT=unknown
|
||||
LABEL org.opencontainers.image.revision="${GIT_COMMIT}"
|
||||
WORKDIR /
|
||||
COPY --from=builder /workspace/manager .
|
||||
USER 65532:65532
|
||||
|
||||
15
Makefile
15
Makefile
@@ -2,6 +2,9 @@
|
||||
IMG ?= controller:latest
|
||||
# YEAR defines the year value used for substituting the YEAR placeholder in the boilerplate header.
|
||||
YEAR ?= $(shell date +%Y)
|
||||
# GIT_COMMIT is baked into the image (-ldflags in the Dockerfile); -dirty
|
||||
# covers staged and untracked changes too, which `git diff --quiet` misses.
|
||||
GIT_COMMIT ?= $(shell git rev-parse --short=12 HEAD 2>/dev/null || echo unknown)$(shell test -z "$$(git status --porcelain 2>/dev/null)" || echo -dirty)
|
||||
|
||||
# Get the currently used golang install path (in GOPATH/bin, unless GOBIN is set)
|
||||
ifeq (,$(shell go env GOBIN))
|
||||
@@ -61,7 +64,7 @@ vet: ## Run go vet against code.
|
||||
|
||||
.PHONY: test
|
||||
test: manifests generate fmt vet setup-envtest ## Run tests.
|
||||
KUBEBUILDER_ASSETS="$(shell "$(ENVTEST)" use $(ENVTEST_K8S_VERSION) --bin-dir "$(LOCALBIN)" -p path)" go test $$(go list ./... | grep -v /e2e) -coverprofile cover.out
|
||||
KUBEBUILDER_ASSETS="$(shell "$(ENVTEST)" use $(ENVTEST_K8S_VERSION) --bin-dir "$(LOCALBIN)" -p path)" go test -race $$(go list ./... | grep -v /e2e) -coverprofile cover.out
|
||||
|
||||
# TODO(user): To use a different vendor for e2e tests, modify the setup under 'tests/e2e'.
|
||||
# The default setup assumes Kind is pre-installed and builds/loads the Manager Docker image locally.
|
||||
@@ -110,18 +113,22 @@ lint-config: golangci-lint ## Verify golangci-lint linter configuration
|
||||
|
||||
.PHONY: build
|
||||
build: manifests generate fmt vet ## Build manager binary.
|
||||
go build -o bin/manager cmd/main.go
|
||||
go build -o bin/manager ./cmd
|
||||
|
||||
.PHONY: run
|
||||
run: manifests generate fmt vet ## Run a controller from your host.
|
||||
go run ./cmd/main.go
|
||||
|
||||
.PHONY: run-dev
|
||||
run-dev: manifests generate fmt vet ## Run locally against the current kubeconfig with the kubernetes-pod provider.
|
||||
go run ./cmd/main.go --providers-config hack/providers-dev.yaml --metrics-bind-address :8080 --metrics-secure=false
|
||||
|
||||
# If you wish to build the manager image targeting other platforms you can use the --platform flag.
|
||||
# (i.e. docker build --platform linux/arm64). However, you must enable docker buildKit for it.
|
||||
# More info: https://docs.docker.com/develop/develop-images/build_enhancements/
|
||||
.PHONY: docker-build
|
||||
docker-build: ## Build docker image with the manager.
|
||||
$(CONTAINER_TOOL) build -t ${IMG} .
|
||||
$(CONTAINER_TOOL) build --build-arg GIT_COMMIT=$(GIT_COMMIT) -t ${IMG} .
|
||||
|
||||
.PHONY: docker-push
|
||||
docker-push: ## Push docker image with the manager.
|
||||
@@ -140,7 +147,7 @@ docker-buildx: ## Build and push docker image for the manager for cross-platform
|
||||
sed -e '1 s/\(^FROM\)/FROM --platform=\$$\{BUILDPLATFORM\}/; t' -e ' 1,// s//FROM --platform=\$$\{BUILDPLATFORM\}/' Dockerfile > Dockerfile.cross
|
||||
- $(CONTAINER_TOOL) buildx create --name egress-proxies-operator-builder
|
||||
$(CONTAINER_TOOL) buildx use egress-proxies-operator-builder
|
||||
- $(CONTAINER_TOOL) buildx build --push --platform=$(PLATFORMS) --tag ${IMG} -f Dockerfile.cross .
|
||||
- $(CONTAINER_TOOL) buildx build --push --platform=$(PLATFORMS) --build-arg GIT_COMMIT=$(GIT_COMMIT) --tag ${IMG} -f Dockerfile.cross .
|
||||
- $(CONTAINER_TOOL) buildx rm egress-proxies-operator-builder
|
||||
rm Dockerfile.cross
|
||||
|
||||
|
||||
292
README.md
292
README.md
@@ -1,135 +1,225 @@
|
||||
# egress-proxies-operator
|
||||
// TODO(user): Add simple overview of use/purpose
|
||||
|
||||
## Description
|
||||
// TODO(user): An in-depth paragraph about your project and overview of use
|
||||
A Kubernetes operator that manages a fleet of HTTP egress proxies for
|
||||
crawling: each proxy is a `Proxy` custom resource that the operator
|
||||
provisions (or merely tracks), actively health-checks **through the proxy
|
||||
itself**, and hands out to crawler clients via an HTTP list/lease API.
|
||||
|
||||
## Getting Started
|
||||
## Architecture in 60 seconds
|
||||
|
||||
### Prerequisites
|
||||
- go version v1.24.6+
|
||||
- docker version 17.03+.
|
||||
- kubectl version v1.11.3+.
|
||||
- Access to a Kubernetes v1.11.3+ cluster.
|
||||
- **`Proxy` CRD** (`crawl.example.com/v1alpha1`, namespaced, `kubectl get px`):
|
||||
`Managed` proxies are provisioned by a configured provider; `External`
|
||||
proxies exist elsewhere and are only tracked and health-checked.
|
||||
- **Reconciler** — a crash-safe state machine: every reconcile derives one
|
||||
action from (spec, status, provider Get). Proxies are **immutable
|
||||
cattle**: any meaningful spec change (placement, cloud-init, port)
|
||||
deletes and recreates the VM — never in-place mutation.
|
||||
- **Providers** behind one minimal interface: `kubernetes` (a real Squid
|
||||
pod in this cluster — local dev/CI) and `gcp` (Compute Engine VMs with
|
||||
ephemeral external IPs — the real egress fleet). Config is a YAML file
|
||||
(`--providers-config`) with named instances (`gcp-eu`, `gcp-us`, ...).
|
||||
- **Health engine** probes every proxy by fetching a URL *through* it (a
|
||||
real CONNECT tunnel — a proxy that accepts TCP but can't egress goes
|
||||
Unhealthy), with threshold logic and transition-only status writes.
|
||||
- **Discovery API** (`:8090`): list healthy proxies filtered by
|
||||
attributes, lease one (least-loaded, TTL-based), release, and report
|
||||
rate-limiting — reports put the proxy in a per-target cooldown.
|
||||
- **Orphan GC** sweeps each provider for tagged instances whose owning CR
|
||||
is gone — the safety net for crashes mid-create.
|
||||
|
||||
### To Deploy on the cluster
|
||||
**Build and push your image to the location specified by `IMG`:**
|
||||
Details, diagrams, and recorded design decisions: [docs/architecture.md](docs/architecture.md).
|
||||
|
||||
## Quickstart on kind (~5 minutes)
|
||||
|
||||
Requires: kind, kubectl, docker, Go 1.26, jq (optional). The
|
||||
kubernetes-pod provider needs no cloud account — proxies are real
|
||||
`ubuntu/squid` pods in the kind cluster itself.
|
||||
|
||||
The operator runs **in-cluster** for this quickstart. (Running it on your
|
||||
laptop with `make run-dev` provisions pods fine, but the health probe then
|
||||
originates on your machine, which cannot reach kind's pod IPs — the proxy
|
||||
would sit at `Unhealthy` forever. In-cluster, probes run where the pod
|
||||
network is routable.)
|
||||
|
||||
```sh
|
||||
make docker-build docker-push IMG=<some-registry>/egress-proxies-operator:tag
|
||||
kind create cluster --name proxy-operator-demo
|
||||
make install # install the CRD
|
||||
make docker-build IMG=egress-proxies-operator:dev
|
||||
kind load docker-image egress-proxies-operator:dev --name proxy-operator-demo
|
||||
make deploy IMG=egress-proxies-operator:dev
|
||||
kubectl -n egress-proxies-operator-system rollout status deploy/egress-proxies-operator-controller-manager
|
||||
```
|
||||
|
||||
**NOTE:** This image ought to be published in the personal registry you specified.
|
||||
And it is required to have access to pull the image from the working environment.
|
||||
Make sure you have the proper permission to the registry if the above commands don’t work.
|
||||
|
||||
**Install the CRDs into the cluster:**
|
||||
Create a proxy and watch it come up:
|
||||
|
||||
```sh
|
||||
make install
|
||||
kubectl apply -f config/samples/proxy_kubernetes.yaml
|
||||
kubectl get px -w
|
||||
# NAME MODE PROVIDER PHASE IP HEALTHY
|
||||
# proxy-kubernetes-sample Managed kubernetes Ready 10.244.x.x True
|
||||
```
|
||||
|
||||
**Deploy the Manager to the cluster with the image specified by `IMG`:**
|
||||
Once it's `Ready`, port-forward the discovery API and use it (full
|
||||
reference with schemas and error codes: [docs/api.md](docs/api.md)):
|
||||
|
||||
```sh
|
||||
make deploy IMG=<some-registry>/egress-proxies-operator:tag
|
||||
kubectl -n egress-proxies-operator-system port-forward \
|
||||
svc/egress-proxies-operator-controller-manager-discovery-service 8090:8090 &
|
||||
```
|
||||
|
||||
> **NOTE**: If you encounter RBAC errors, you may need to grant yourself cluster-admin
|
||||
privileges or be logged in as admin.
|
||||
|
||||
**Create instances of your solution**
|
||||
You can apply the samples (examples) from the config/sample:
|
||||
|
||||
```sh
|
||||
kubectl apply -k config/samples/
|
||||
# List healthy proxies
|
||||
curl -s 'localhost:8090/v1/proxies?healthy=true' | jq
|
||||
|
||||
# Lease one (5-minute TTL)
|
||||
curl -s -XPOST localhost:8090/v1/leases \
|
||||
-d '{"selector":{"geo":"local"},"ttlSeconds":300}' | jq
|
||||
# → {"leaseID":"...", "proxy":{"id":"default/proxy-kubernetes-sample", "ip":..., ...}}
|
||||
|
||||
# Actually crawl through it (from inside the cluster, or port-forward the pod)
|
||||
# curl -x http://<proxy-ip>:3128 https://example.com
|
||||
|
||||
# Report the proxy got rate-limited by a site → 15-minute cooldown for that target
|
||||
curl -s -XPOST localhost:8090/v1/leases/<leaseID>/report \
|
||||
-d '{"result":"rate_limited","target":"example.com"}'
|
||||
|
||||
# Release early (idempotent — 204 both times)
|
||||
curl -si -XDELETE localhost:8090/v1/leases/<leaseID>
|
||||
```
|
||||
|
||||
>**NOTE**: Ensure that the samples has default values to test it out.
|
||||
|
||||
### To Uninstall
|
||||
**Delete the instances (CRs) from the cluster:**
|
||||
Tear down:
|
||||
|
||||
```sh
|
||||
kubectl delete -k config/samples/
|
||||
kubectl delete -f config/samples/proxy_kubernetes.yaml # finalizer deletes the pod
|
||||
kind delete cluster --name proxy-operator-demo
|
||||
```
|
||||
|
||||
**Delete the APIs(CRDs) from the cluster:**
|
||||
## Deploying in-cluster
|
||||
|
||||
```sh
|
||||
make uninstall
|
||||
make docker-build IMG=<registry>/egress-proxies-operator:dev
|
||||
make deploy IMG=<registry>/egress-proxies-operator:dev
|
||||
```
|
||||
|
||||
**UnDeploy the controller from the cluster:**
|
||||
- Provider config comes from the `providers-config` ConfigMap
|
||||
([config/manager/providers_config.yaml](config/manager/providers_config.yaml));
|
||||
the default ships only the kubernetes provider.
|
||||
- The discovery API is exposed by the
|
||||
`controller-manager-discovery-service` Service on port 8090.
|
||||
- Auth: create the token Secret, or the API serves **unauthenticated**
|
||||
(it warns loudly at startup):
|
||||
|
||||
```sh
|
||||
kubectl -n egress-proxies-operator-system create secret generic discovery-token \
|
||||
--from-literal=token="$(openssl rand -hex 24)"
|
||||
```
|
||||
|
||||
## GCP setup
|
||||
|
||||
1. Add a `gcp` entry to the providers config (see
|
||||
[config/samples/providers-config.yaml](config/samples/providers-config.yaml)) —
|
||||
only `project` is required.
|
||||
2. Credentials are **Application Default Credentials**: workload identity
|
||||
in-cluster, `gcloud auth application-default login` locally. No
|
||||
key-file plumbing exists.
|
||||
3. The identity needs `roles/compute.instanceAdmin.v1` on the project —
|
||||
plus `roles/iam.serviceAccountUser` if instances attach a service
|
||||
account.
|
||||
4. Managed GCP proxies must set all of `placement.zone`,
|
||||
`placement.machineType`, and `placement.image`
|
||||
(see [config/samples/proxy_gcp.yaml](config/samples/proxy_gcp.yaml),
|
||||
which also installs Squid via cloud-init). A missing field fails the
|
||||
Proxy with a message naming it.
|
||||
|
||||
Cloud-init from a Secret: the Secret **must** carry the label
|
||||
`crawl.example.com/cloud-init: "true"` — the operator's cache only holds
|
||||
labelled Secrets, so an unlabelled one is invisible (the Proxy reports
|
||||
`CloudInitError`). Rotating the Secret's content triggers VM replacement.
|
||||
|
||||
## Caveats — read these two
|
||||
|
||||
**Changing a proxy changes its IP.** Proxies are immutable cattle: editing
|
||||
`placement`, `cloudInit` (or rotating its Secret), or `port` deletes the
|
||||
VM and creates a replacement with the **same name but a new IP**. Clients
|
||||
discover the new address via the discovery API; anything that pinned the
|
||||
old IP breaks by design.
|
||||
|
||||
**Operator restart drops all leases and cooldowns.** Lease state is
|
||||
in-memory (`replicas: 1` accordingly). Clients must tolerate a lease
|
||||
vanishing — requests through the proxy keep working; they just re-lease.
|
||||
The lease store sits behind an interface so a persistent backend can
|
||||
replace it without touching the API handlers.
|
||||
|
||||
Smaller notes:
|
||||
|
||||
- `status.lastHealthCheckTime` is the time of the last *status-affecting*
|
||||
probe, not the most recent probe — status writes are transition-only by
|
||||
design. True probe recency lives in the metrics
|
||||
(`proxy_operator_healthcheck_*`).
|
||||
- The discovery API is served by every replica but is not leader-elected;
|
||||
the operator ships with `replicas: 1` (see the lease caveat above).
|
||||
|
||||
## Version pins
|
||||
|
||||
Built and verified against the spec's pins with **no substitutions
|
||||
needed**: Go 1.26, kubebuilder v4.15.0, controller-runtime v0.24.1,
|
||||
k8s.io/* v0.36.3 (Kubernetes 1.36 API level), controller-tools v0.21.0,
|
||||
cloud.google.com/go/compute v1.65.0. envtest uses the 1.36.2 binary
|
||||
bundle (the latest 1.36 patch with published binaries — do not "fix" the
|
||||
Makefile's derived version to 1.36.3, which has none).
|
||||
|
||||
## Gitea CI
|
||||
|
||||
[.gitea/workflows/build.yaml](.gitea/workflows/build.yaml) builds the
|
||||
manager image and pushes it to this Gitea instance's container registry.
|
||||
It runs on **any tag push** or manually via **Run workflow** (with a `tag`
|
||||
input) — never on branch pushes. A lightweight `check` job (`go vet`,
|
||||
`go build`, `go test -short`) gates the build.
|
||||
|
||||
Every build pushes two tags to
|
||||
`gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator`:
|
||||
|
||||
- the human tag (the git tag, or the dispatch input), and
|
||||
- an immutable `sha-<12-char-commit>` tag — pin deployments to this one.
|
||||
|
||||
`:latest` is additionally updated on real tag pushes only, so a manual
|
||||
dispatch of an old ref can never clobber it. The commit is baked into the
|
||||
binary (`internal/version.Commit`) via the `GIT_COMMIT` build arg.
|
||||
|
||||
### Mandatory Gitea secrets
|
||||
|
||||
Set under **Settings → Actions → Secrets** in this repo:
|
||||
|
||||
| Secret | Required by | What it is |
|
||||
| ---------------- | ----------------------------- | ---------------------------------------- |
|
||||
| `REGISTRY_TOKEN` | `build.yaml` (registry login) | Gitea PAT with the `write:package` scope |
|
||||
|
||||
The token is paired with `${{ github.actor }}` as the username, so it
|
||||
must belong to the user triggering the workflow — same convention as the
|
||||
other projects on this instance.
|
||||
|
||||
Without `REGISTRY_TOKEN` the `check` job still passes but the build job
|
||||
fails at the `docker login` step. No other secrets are needed — the
|
||||
workflow does not deploy anywhere.
|
||||
|
||||
## Development
|
||||
|
||||
```sh
|
||||
make undeploy
|
||||
make test # unit + envtest suites, with -race (sets up envtest binaries itself)
|
||||
go test -short ./... # skip the envtest suite
|
||||
make run-dev # run against the current kubeconfig context
|
||||
```
|
||||
|
||||
## Project Distribution
|
||||
`make run-dev` is for iterating on the operator itself: provisioning,
|
||||
replacement, the discovery API, and External proxies all work from your
|
||||
laptop. Health checks against in-cluster pods do **not** (see the
|
||||
quickstart note) — use the in-cluster deploy to see a kubernetes-provider
|
||||
proxy go `Ready`.
|
||||
|
||||
Following the options to release and provide this solution to the users.
|
||||
|
||||
### By providing a bundle with all YAML files
|
||||
|
||||
1. Build the installer for the image built and published in the registry:
|
||||
|
||||
```sh
|
||||
make build-installer IMG=<some-registry>/egress-proxies-operator:tag
|
||||
```
|
||||
|
||||
**NOTE:** The makefile target mentioned above generates an 'install.yaml'
|
||||
file in the dist directory. This file contains all the resources built
|
||||
with Kustomize, which are necessary to install this project without its
|
||||
dependencies.
|
||||
|
||||
2. Using the installer
|
||||
|
||||
Users can just run 'kubectl apply -f <URL for YAML BUNDLE>' to install
|
||||
the project, i.e.:
|
||||
|
||||
```sh
|
||||
kubectl apply -f https://raw.githubusercontent.com/<org>/egress-proxies-operator/<tag or branch>/dist/install.yaml
|
||||
```
|
||||
|
||||
### By providing a Helm Chart
|
||||
|
||||
1. Build the chart using the optional helm plugin
|
||||
|
||||
```sh
|
||||
kubebuilder edit --plugins=helm/v2-alpha
|
||||
```
|
||||
|
||||
2. See that a chart was generated under 'dist/chart', and users
|
||||
can obtain this solution from there.
|
||||
|
||||
**NOTE:** If you change the project, you need to update the Helm Chart
|
||||
using the same command above to sync the latest changes. Furthermore,
|
||||
if you create webhooks, you need to use the above command with
|
||||
the '--force' flag and manually ensure that any custom configuration
|
||||
previously added to 'dist/chart/values.yaml' or 'dist/chart/manager/manager.yaml'
|
||||
is manually re-applied afterwards.
|
||||
|
||||
## Contributing
|
||||
// TODO(user): Add detailed information on how you would like others to contribute to this project
|
||||
|
||||
**NOTE:** Run `make help` for more information on all potential `make` targets
|
||||
|
||||
More information can be found via the [Kubebuilder Documentation](https://book.kubebuilder.io/introduction.html)
|
||||
|
||||
## License
|
||||
|
||||
Copyright 2026.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
The full test inventory — what each suite covers, the deliberate gaps,
|
||||
and the manual kind verification procedure — is in
|
||||
[docs/testing.md](docs/testing.md).
|
||||
|
||||
Project layout, reconcile-loop diagrams, and the decision log are in
|
||||
[docs/architecture.md](docs/architecture.md); the build history is in
|
||||
[docs/plans-executions/](docs/plans-executions/).
|
||||
|
||||
87
api/v1alpha1/helpers.go
Normal file
87
api/v1alpha1/helpers.go
Normal file
@@ -0,0 +1,87 @@
|
||||
/*
|
||||
Copyright 2026.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
*/
|
||||
|
||||
package v1alpha1
|
||||
|
||||
// EffectivePort returns the port a client should use to reach the proxy:
|
||||
// spec.endpoint.port for External proxies, spec.port for Managed proxies.
|
||||
// Falls back to DefaultPort if the relevant field is unset, so callers that
|
||||
// bypass CRD structural defaulting (unit tests, fake clients) still get a
|
||||
// sane value.
|
||||
func (p *Proxy) EffectivePort() int32 {
|
||||
if p.Spec.Mode == ModeExternal && p.Spec.Endpoint != nil {
|
||||
if p.Spec.Endpoint.Port != 0 {
|
||||
return p.Spec.Endpoint.Port
|
||||
}
|
||||
return DefaultPort
|
||||
}
|
||||
if p.Spec.Port != 0 {
|
||||
return p.Spec.Port
|
||||
}
|
||||
return DefaultPort
|
||||
}
|
||||
|
||||
// EffectiveHost returns the host a client should use to reach the proxy:
|
||||
// spec.endpoint.host for External proxies, status.ip for Managed proxies
|
||||
// (populated once the VM is running).
|
||||
func (p *Proxy) EffectiveHost() string {
|
||||
if p.Spec.Mode == ModeExternal && p.Spec.Endpoint != nil {
|
||||
return p.Spec.Endpoint.Host
|
||||
}
|
||||
return p.Status.IP
|
||||
}
|
||||
|
||||
// HealthCheckOrDefault returns spec.healthCheck with every unset field
|
||||
// filled from its default. CRD structural defaulting (the default={} marker
|
||||
// on ProxySpec.HealthCheck) already does this for objects that went through
|
||||
// the API server; this is for callers that didn't (unit tests, fake
|
||||
// clients, or a Proxy constructed directly in Go).
|
||||
func (p *Proxy) HealthCheckOrDefault() HealthCheckSpec {
|
||||
var hc HealthCheckSpec
|
||||
if p.Spec.HealthCheck != nil {
|
||||
hc = *p.Spec.HealthCheck
|
||||
}
|
||||
if hc.ProbeURL == "" {
|
||||
hc.ProbeURL = DefaultProbeURL
|
||||
}
|
||||
if hc.IntervalSeconds == 0 {
|
||||
hc.IntervalSeconds = DefaultHealthCheckIntervalSeconds
|
||||
}
|
||||
if hc.TimeoutSeconds == 0 {
|
||||
hc.TimeoutSeconds = DefaultHealthCheckTimeoutSeconds
|
||||
}
|
||||
if hc.FailureThreshold == 0 {
|
||||
hc.FailureThreshold = DefaultFailureThreshold
|
||||
}
|
||||
if hc.SuccessThreshold == 0 {
|
||||
hc.SuccessThreshold = DefaultSuccessThreshold
|
||||
}
|
||||
if len(hc.ExpectedStatusCodes) == 0 {
|
||||
hc.ExpectedStatusCodes = append([]int32(nil), DefaultExpectedStatusCodes...)
|
||||
}
|
||||
return hc
|
||||
}
|
||||
|
||||
// MaxLeasesOrDefault returns spec.maxLeases, or DefaultMaxLeases if unset.
|
||||
// spec.maxLeases is a pointer specifically so an explicit 0 (unleasable) is
|
||||
// distinguishable from "unset" and survives Go round-trips; this helper
|
||||
// preserves that distinction.
|
||||
func (p *Proxy) MaxLeasesOrDefault() int32 {
|
||||
if p.Spec.MaxLeases != nil {
|
||||
return *p.Spec.MaxLeases
|
||||
}
|
||||
return DefaultMaxLeases
|
||||
}
|
||||
190
api/v1alpha1/helpers_test.go
Normal file
190
api/v1alpha1/helpers_test.go
Normal file
@@ -0,0 +1,190 @@
|
||||
/*
|
||||
Copyright 2026.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
*/
|
||||
|
||||
package v1alpha1
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestEffectivePort(t *testing.T) {
|
||||
t.Parallel()
|
||||
tests := []struct {
|
||||
name string
|
||||
spec ProxySpec
|
||||
want int32
|
||||
}{
|
||||
{
|
||||
name: "managed with explicit port",
|
||||
spec: ProxySpec{Mode: ModeManaged, Port: 8080},
|
||||
want: 8080,
|
||||
},
|
||||
{
|
||||
name: "managed with unset port falls back to default",
|
||||
spec: ProxySpec{Mode: ModeManaged},
|
||||
want: DefaultPort,
|
||||
},
|
||||
{
|
||||
name: "external with explicit endpoint port",
|
||||
spec: ProxySpec{Mode: ModeExternal, Endpoint: &EndpointSpec{Host: "1.2.3.4", Port: 9999}},
|
||||
want: 9999,
|
||||
},
|
||||
{
|
||||
name: "external with unset endpoint port falls back to default",
|
||||
spec: ProxySpec{Mode: ModeExternal, Endpoint: &EndpointSpec{Host: "1.2.3.4"}},
|
||||
want: DefaultPort,
|
||||
},
|
||||
{
|
||||
name: "external with nil endpoint falls back to spec.port",
|
||||
spec: ProxySpec{Mode: ModeExternal, Port: 3000},
|
||||
want: 3000,
|
||||
},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := &Proxy{Spec: tc.spec}
|
||||
if got := p.EffectivePort(); got != tc.want {
|
||||
t.Errorf("EffectivePort() = %d, want %d", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestEffectiveHost(t *testing.T) {
|
||||
t.Parallel()
|
||||
tests := []struct {
|
||||
name string
|
||||
spec ProxySpec
|
||||
status ProxyStatus
|
||||
want string
|
||||
}{
|
||||
{
|
||||
name: "managed uses status.ip",
|
||||
spec: ProxySpec{Mode: ModeManaged},
|
||||
status: ProxyStatus{IP: "10.0.0.5"},
|
||||
want: "10.0.0.5",
|
||||
},
|
||||
{
|
||||
name: "external uses endpoint.host",
|
||||
spec: ProxySpec{Mode: ModeExternal, Endpoint: &EndpointSpec{Host: "proxy.example.com"}},
|
||||
want: "proxy.example.com",
|
||||
},
|
||||
{
|
||||
name: "external with nil endpoint falls back to status.ip",
|
||||
spec: ProxySpec{Mode: ModeExternal},
|
||||
status: ProxyStatus{IP: "10.0.0.6"},
|
||||
want: "10.0.0.6",
|
||||
},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := &Proxy{Spec: tc.spec, Status: tc.status}
|
||||
if got := p.EffectiveHost(); got != tc.want {
|
||||
t.Errorf("EffectiveHost() = %q, want %q", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestHealthCheckOrDefault(t *testing.T) {
|
||||
t.Parallel()
|
||||
fullDefault := HealthCheckSpec{
|
||||
ProbeURL: DefaultProbeURL,
|
||||
IntervalSeconds: DefaultHealthCheckIntervalSeconds,
|
||||
TimeoutSeconds: DefaultHealthCheckTimeoutSeconds,
|
||||
FailureThreshold: DefaultFailureThreshold,
|
||||
SuccessThreshold: DefaultSuccessThreshold,
|
||||
ExpectedStatusCodes: []int32{200, 204},
|
||||
}
|
||||
|
||||
t.Run("nil healthCheck returns full default", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := &Proxy{Spec: ProxySpec{}}
|
||||
got := p.HealthCheckOrDefault()
|
||||
if !reflect.DeepEqual(got, fullDefault) {
|
||||
t.Errorf("HealthCheckOrDefault() = %+v, want %+v", got, fullDefault)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("partial healthCheck fills only unset fields", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := &Proxy{Spec: ProxySpec{HealthCheck: &HealthCheckSpec{
|
||||
ProbeURL: "http://internal/probe",
|
||||
FailureThreshold: 7,
|
||||
}}}
|
||||
got := p.HealthCheckOrDefault()
|
||||
want := fullDefault
|
||||
want.ProbeURL = "http://internal/probe"
|
||||
want.FailureThreshold = 7
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("HealthCheckOrDefault() = %+v, want %+v", got, want)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("fully set healthCheck passes through unchanged", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
custom := HealthCheckSpec{
|
||||
ProbeURL: "http://internal/probe",
|
||||
IntervalSeconds: 10,
|
||||
TimeoutSeconds: 2,
|
||||
FailureThreshold: 5,
|
||||
SuccessThreshold: 2,
|
||||
ExpectedStatusCodes: []int32{200},
|
||||
}
|
||||
p := &Proxy{Spec: ProxySpec{HealthCheck: &custom}}
|
||||
got := p.HealthCheckOrDefault()
|
||||
if !reflect.DeepEqual(got, custom) {
|
||||
t.Errorf("HealthCheckOrDefault() = %+v, want %+v", got, custom)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("does not mutate the original spec", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
hc := &HealthCheckSpec{ProbeURL: "http://internal/probe"}
|
||||
p := &Proxy{Spec: ProxySpec{HealthCheck: hc}}
|
||||
_ = p.HealthCheckOrDefault()
|
||||
if hc.IntervalSeconds != 0 {
|
||||
t.Errorf("original HealthCheckSpec was mutated: IntervalSeconds = %d, want 0", hc.IntervalSeconds)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestMaxLeasesOrDefault(t *testing.T) {
|
||||
t.Parallel()
|
||||
zero := int32(0)
|
||||
seven := int32(7)
|
||||
tests := []struct {
|
||||
name string
|
||||
spec ProxySpec
|
||||
want int32
|
||||
}{
|
||||
{name: "unset falls back to default", spec: ProxySpec{}, want: DefaultMaxLeases},
|
||||
{name: "explicit zero is preserved (unleasable)", spec: ProxySpec{MaxLeases: &zero}, want: 0},
|
||||
{name: "explicit non-zero is preserved", spec: ProxySpec{MaxLeases: &seven}, want: 7},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := &Proxy{Spec: tc.spec}
|
||||
if got := p.MaxLeasesOrDefault(); got != tc.want {
|
||||
t.Errorf("MaxLeasesOrDefault() = %d, want %d", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -21,46 +21,308 @@ import (
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
)
|
||||
|
||||
// EDIT THIS FILE! THIS IS SCAFFOLDING FOR YOU TO OWN!
|
||||
// NOTE: json tags are required. Any new fields you add must have json tags for the fields to be serialized.
|
||||
// ProvisioningMode selects who owns the lifecycle of the proxy VM.
|
||||
type ProvisioningMode string
|
||||
|
||||
// ProxySpec defines the desired state of Proxy
|
||||
type ProxySpec struct {
|
||||
// INSERT ADDITIONAL SPEC FIELDS - desired state of cluster
|
||||
// Important: Run "make" to regenerate code after modifying this file
|
||||
// The following markers will use OpenAPI v3 schema to validate the value
|
||||
// More info: https://book.kubebuilder.io/reference/markers/crd-validation.html
|
||||
const (
|
||||
// ModeManaged means the operator creates, monitors, and deletes the VM.
|
||||
ModeManaged ProvisioningMode = "Managed"
|
||||
// ModeExternal means the VM exists outside the operator's control; the
|
||||
// operator only tracks and healthchecks it.
|
||||
ModeExternal ProvisioningMode = "External"
|
||||
)
|
||||
|
||||
// foo is an example field of Proxy. Edit proxy_types.go to remove/update
|
||||
// ProxyPhase is a high-level, human-readable summary of status.conditions.
|
||||
type ProxyPhase string
|
||||
|
||||
const (
|
||||
PhasePending ProxyPhase = "Pending"
|
||||
PhaseProvisioning ProxyPhase = "Provisioning"
|
||||
PhaseReady ProxyPhase = "Ready"
|
||||
PhaseUnhealthy ProxyPhase = "Unhealthy"
|
||||
PhaseDeleting ProxyPhase = "Deleting"
|
||||
PhaseFailed ProxyPhase = "Failed"
|
||||
)
|
||||
|
||||
const (
|
||||
// ConditionProvisioned reflects the state of the underlying VM (Managed)
|
||||
// or endpoint (External).
|
||||
ConditionProvisioned = "Provisioned"
|
||||
// ConditionHealthy reflects the result of the through-the-proxy healthcheck.
|
||||
ConditionHealthy = "Healthy"
|
||||
|
||||
// FinalizerName is set on Managed proxies so deletion can clean up the
|
||||
// provider-side VM before the CR is removed. External proxies never get
|
||||
// this finalizer.
|
||||
FinalizerName = "crawl.example.com/proxy-cleanup"
|
||||
|
||||
// AnnotationSpecHash stores the hash of the replacement-triggering spec
|
||||
// fields (placement, resolved cloud-init, port) as of the last successful
|
||||
// provision. A mismatch against the freshly computed hash means the VM
|
||||
// must be replaced.
|
||||
AnnotationSpecHash = "crawl.example.com/spec-hash"
|
||||
|
||||
// LabelCloudInit must be set (to "true") on every Secret referenced by
|
||||
// spec.cloudInit.secretRef: the manager's cache only holds Secrets
|
||||
// carrying this label, so an unlabelled Secret is invisible to the
|
||||
// operator — both to the resolve step and to the rotation watch.
|
||||
LabelCloudInit = "crawl.example.com/cloud-init"
|
||||
)
|
||||
|
||||
// Defaults, applied both by CRD structural defaulting (kubebuilder:default
|
||||
// markers below) and by the OrDefault helpers for callers that bypass the
|
||||
// API server (unit tests, fake clients).
|
||||
const (
|
||||
DefaultPort int32 = 3128
|
||||
DefaultMaxLeases int32 = 5
|
||||
DefaultProbeURL string = "https://www.gstatic.com/generate_204"
|
||||
DefaultHealthCheckIntervalSeconds int32 = 30
|
||||
DefaultHealthCheckTimeoutSeconds int32 = 5
|
||||
DefaultFailureThreshold int32 = 3
|
||||
DefaultSuccessThreshold int32 = 1
|
||||
DefaultCloudInitSecretKey string = "user-data"
|
||||
)
|
||||
|
||||
// DefaultExpectedStatusCodes is the default set of HTTP status codes a
|
||||
// healthcheck probe treats as success.
|
||||
var DefaultExpectedStatusCodes = []int32{200, 204}
|
||||
|
||||
// PlacementSpec is provider-opaque placement/size configuration. It is kept
|
||||
// as a small typed struct rather than map[string]string; providers ignore
|
||||
// fields that don't apply to them.
|
||||
type PlacementSpec struct {
|
||||
// region is the provider's region identifier (e.g. "europe-west1").
|
||||
// +optional
|
||||
Foo *string `json:"foo,omitempty"`
|
||||
Region string `json:"region,omitempty"`
|
||||
|
||||
// zone is the provider's zone identifier (e.g. "europe-west1-b").
|
||||
// +optional
|
||||
Zone string `json:"zone,omitempty"`
|
||||
|
||||
// machineType is the provider's machine/instance type (e.g. "e2-micro").
|
||||
// +optional
|
||||
MachineType string `json:"machineType,omitempty"`
|
||||
|
||||
// image is the boot image reference.
|
||||
// +optional
|
||||
Image string `json:"image,omitempty"`
|
||||
}
|
||||
|
||||
// SecretKeySelector references a key within a Secret in the same namespace.
|
||||
type SecretKeySelector struct {
|
||||
// name is the Secret's name.
|
||||
// +kubebuilder:validation:MinLength=1
|
||||
Name string `json:"name"`
|
||||
|
||||
// key is the data key holding the cloud-init user-data.
|
||||
// +kubebuilder:default=user-data
|
||||
// +optional
|
||||
Key string `json:"key,omitempty"`
|
||||
}
|
||||
|
||||
// CloudInitSpec supplies cloud-init user-data either inline or from a
|
||||
// Secret. Exactly one of Inline or SecretRef must be set. Changing the
|
||||
// resolved content on a Managed proxy triggers replacement.
|
||||
//
|
||||
// +kubebuilder:validation:XValidation:rule="has(self.inline) != has(self.secretRef)",message="exactly one of inline or secretRef must be set"
|
||||
type CloudInitSpec struct {
|
||||
// inline is the literal cloud-init user-data.
|
||||
// +kubebuilder:validation:MinLength=1
|
||||
// +kubebuilder:validation:MaxLength=262144
|
||||
// +optional
|
||||
Inline string `json:"inline,omitempty"`
|
||||
|
||||
// secretRef points at a Secret key holding the cloud-init user-data.
|
||||
// +optional
|
||||
SecretRef *SecretKeySelector `json:"secretRef,omitempty"`
|
||||
}
|
||||
|
||||
// EndpointSpec identifies an External proxy's network location.
|
||||
type EndpointSpec struct {
|
||||
// host is the proxy's hostname or IP address.
|
||||
// +kubebuilder:validation:MinLength=1
|
||||
Host string `json:"host"`
|
||||
|
||||
// port is the port the proxy listens on.
|
||||
// +kubebuilder:default=3128
|
||||
// +kubebuilder:validation:Minimum=1
|
||||
// +kubebuilder:validation:Maximum=65535
|
||||
// +optional
|
||||
Port int32 `json:"port,omitempty"`
|
||||
}
|
||||
|
||||
// HealthCheckSpec configures the through-the-proxy healthcheck. All fields
|
||||
// are defaulted, both by the CRD (kubebuilder:default markers, activated by
|
||||
// the default={} marker on ProxySpec.HealthCheck) and by
|
||||
// Proxy.HealthCheckOrDefault for callers that bypass the API server.
|
||||
type HealthCheckSpec struct {
|
||||
// probeURL is fetched through the proxy on every probe.
|
||||
// +kubebuilder:default="https://www.gstatic.com/generate_204"
|
||||
// +kubebuilder:validation:MinLength=1
|
||||
// +optional
|
||||
ProbeURL string `json:"probeURL,omitempty"`
|
||||
|
||||
// intervalSeconds is the time between probes for a given proxy.
|
||||
// +kubebuilder:default=30
|
||||
// +kubebuilder:validation:Minimum=5
|
||||
// +optional
|
||||
IntervalSeconds int32 `json:"intervalSeconds,omitempty"`
|
||||
|
||||
// timeoutSeconds bounds a single probe, including the CONNECT tunnel
|
||||
// setup and the TLS handshake through it.
|
||||
// +kubebuilder:default=5
|
||||
// +kubebuilder:validation:Minimum=1
|
||||
// +optional
|
||||
TimeoutSeconds int32 `json:"timeoutSeconds,omitempty"`
|
||||
|
||||
// failureThreshold is the number of consecutive failed probes required
|
||||
// to transition Healthy -> False.
|
||||
// +kubebuilder:default=3
|
||||
// +kubebuilder:validation:Minimum=1
|
||||
// +optional
|
||||
FailureThreshold int32 `json:"failureThreshold,omitempty"`
|
||||
|
||||
// successThreshold is the number of consecutive successful probes
|
||||
// required to transition Healthy -> True.
|
||||
// +kubebuilder:default=1
|
||||
// +kubebuilder:validation:Minimum=1
|
||||
// +optional
|
||||
SuccessThreshold int32 `json:"successThreshold,omitempty"`
|
||||
|
||||
// expectedStatusCodes are the HTTP status codes a probe response must
|
||||
// match to count as successful.
|
||||
// +kubebuilder:default={200,204}
|
||||
// +kubebuilder:validation:MaxItems=8
|
||||
// +optional
|
||||
ExpectedStatusCodes []int32 `json:"expectedStatusCodes,omitempty"`
|
||||
}
|
||||
|
||||
// ProxySpec defines the desired state of Proxy.
|
||||
//
|
||||
// Cross-field rules are deliberately split into one XValidation marker per
|
||||
// concern: a rule referencing oldSelf is skipped on CREATE, so an
|
||||
// immutability rule and a required-iff rule must never be combined into one
|
||||
// `&&`-ed expression, or the required-iff half would silently stop applying
|
||||
// on CREATE. The provider-immutability rule uses has(self.x)==has(oldSelf.x)
|
||||
// rather than a field-level self==oldSelf rule because Provider is optional:
|
||||
// a field-level rule does not fire when the field is absent on either side,
|
||||
// which would silently allow adding or removing it after creation.
|
||||
//
|
||||
// +kubebuilder:validation:XValidation:rule="self.mode == oldSelf.mode",message="mode is immutable"
|
||||
// +kubebuilder:validation:XValidation:rule="has(self.provider) == has(oldSelf.provider) && (!has(self.provider) || self.provider == oldSelf.provider)",message="provider is immutable"
|
||||
// +kubebuilder:validation:XValidation:rule="self.mode != 'Managed' || has(self.provider)",message="provider is required when mode is Managed"
|
||||
// +kubebuilder:validation:XValidation:rule="self.mode != 'External' || !has(self.provider)",message="provider must not be set when mode is External"
|
||||
// +kubebuilder:validation:XValidation:rule="self.mode != 'External' || has(self.endpoint)",message="endpoint is required when mode is External"
|
||||
// +kubebuilder:validation:XValidation:rule="self.mode != 'Managed' || !has(self.endpoint)",message="endpoint must not be set when mode is Managed"
|
||||
type ProxySpec struct {
|
||||
// mode selects who owns the VM lifecycle: Managed (operator creates it)
|
||||
// or External (operator only tracks and healthchecks it). Immutable.
|
||||
// +kubebuilder:validation:Enum=Managed;External
|
||||
Mode ProvisioningMode `json:"mode"`
|
||||
|
||||
// provider is the name of a configured provider ("mock", "gcp-eu", ...).
|
||||
// Required iff mode is Managed; must be unset iff mode is External.
|
||||
// Immutable.
|
||||
// +kubebuilder:validation:MinLength=1
|
||||
// +kubebuilder:validation:MaxLength=63
|
||||
// +optional
|
||||
Provider string `json:"provider,omitempty"`
|
||||
|
||||
// placement is provider-opaque placement/size configuration. Only
|
||||
// meaningful for Managed proxies.
|
||||
// +optional
|
||||
Placement *PlacementSpec `json:"placement,omitempty"`
|
||||
|
||||
// cloudInit supplies the VM's cloud-init user-data. Only meaningful for
|
||||
// Managed proxies. Changing the resolved content triggers replacement.
|
||||
// +optional
|
||||
CloudInit *CloudInitSpec `json:"cloudInit,omitempty"`
|
||||
|
||||
// endpoint identifies an External proxy's network location. Required
|
||||
// iff mode is External; must be unset iff mode is Managed.
|
||||
// +optional
|
||||
Endpoint *EndpointSpec `json:"endpoint,omitempty"`
|
||||
|
||||
// port is the port the proxy listens on once the VM is up. Only
|
||||
// meaningful for Managed proxies; External proxies use endpoint.port.
|
||||
// +kubebuilder:default=3128
|
||||
// +kubebuilder:validation:Minimum=1
|
||||
// +kubebuilder:validation:Maximum=65535
|
||||
// +optional
|
||||
Port int32 `json:"port,omitempty"`
|
||||
|
||||
// attributes are selection attributes exposed to the discovery API
|
||||
// (geo, asn, purpose, ...). Deliberately separate from Kubernetes object
|
||||
// labels, which stay an operator implementation concern.
|
||||
// +kubebuilder:validation:MaxProperties=32
|
||||
// +optional
|
||||
Attributes map[string]string `json:"attributes,omitempty"`
|
||||
|
||||
// healthCheck configures the through-the-proxy healthcheck. All nested
|
||||
// fields are defaulted; the default={} marker ensures a proxy that omits
|
||||
// healthCheck entirely still gets every nested default.
|
||||
// +kubebuilder:default={}
|
||||
// +optional
|
||||
HealthCheck *HealthCheckSpec `json:"healthCheck,omitempty"`
|
||||
|
||||
// maxLeases is the maximum number of concurrent leases handed out for
|
||||
// this proxy. 0 means unleasable (list-only visibility). A pointer so an
|
||||
// explicit 0 survives Go round-trips instead of being re-defaulted to 5.
|
||||
// +kubebuilder:default=5
|
||||
// +kubebuilder:validation:Minimum=0
|
||||
// +kubebuilder:validation:Maximum=1000
|
||||
// +optional
|
||||
MaxLeases *int32 `json:"maxLeases,omitempty"`
|
||||
}
|
||||
|
||||
// ProxyStatus defines the observed state of Proxy.
|
||||
type ProxyStatus struct {
|
||||
// INSERT ADDITIONAL STATUS FIELD - define observed state of cluster
|
||||
// Important: Run "make" to regenerate code after modifying this file
|
||||
// phase is a high-level, human-readable summary derived from conditions.
|
||||
// One of Pending, Provisioning, Ready, Unhealthy, Deleting, Failed.
|
||||
// +optional
|
||||
Phase ProxyPhase `json:"phase,omitempty"`
|
||||
|
||||
// For Kubernetes API conventions, see:
|
||||
// https://github.com/kubernetes/community/blob/master/contributors/devel/sig-architecture/api-conventions.md#typical-status-properties
|
||||
// providerID is the opaque cloud resource ID returned by the provider.
|
||||
// Empty for External proxies.
|
||||
// +optional
|
||||
ProviderID string `json:"providerID,omitempty"`
|
||||
|
||||
// ip is the proxy's current IP address: the VM's address for Managed
|
||||
// proxies, or spec.endpoint.host for External proxies.
|
||||
// +optional
|
||||
IP string `json:"ip,omitempty"`
|
||||
|
||||
// conditions represent the current state of the Proxy resource.
|
||||
// Each condition has a unique type and reflects the status of a specific aspect of the resource.
|
||||
//
|
||||
// Standard condition types include:
|
||||
// - "Available": the resource is fully functional
|
||||
// - "Progressing": the resource is being created or updated
|
||||
// - "Degraded": the resource failed to reach or maintain its desired state
|
||||
//
|
||||
// The status of each condition is one of True, False, or Unknown.
|
||||
// Standard types are Provisioned and Healthy.
|
||||
// +listType=map
|
||||
// +listMapKey=type
|
||||
// +optional
|
||||
Conditions []metav1.Condition `json:"conditions,omitempty"`
|
||||
|
||||
// lastHealthCheckTime is the time of the last status-affecting probe,
|
||||
// not the time of the most recent probe — probes that don't change the
|
||||
// Healthy condition or move latency materially don't write status.
|
||||
// +optional
|
||||
LastHealthCheckTime *metav1.Time `json:"lastHealthCheckTime,omitempty"`
|
||||
|
||||
// latencyMillis is the latency of the last status-affecting probe.
|
||||
// +optional
|
||||
LatencyMillis int64 `json:"latencyMillis,omitempty"`
|
||||
|
||||
// observedGeneration is the .metadata.generation last reconciled.
|
||||
// +optional
|
||||
ObservedGeneration int64 `json:"observedGeneration,omitempty"`
|
||||
}
|
||||
|
||||
// +kubebuilder:object:root=true
|
||||
// +kubebuilder:subresource:status
|
||||
// +kubebuilder:resource:shortName=px
|
||||
// +kubebuilder:printcolumn:name="Mode",type=string,JSONPath=`.spec.mode`
|
||||
// +kubebuilder:printcolumn:name="Provider",type=string,JSONPath=`.spec.provider`
|
||||
// +kubebuilder:printcolumn:name="Phase",type=string,JSONPath=`.status.phase`
|
||||
// +kubebuilder:printcolumn:name="IP",type=string,JSONPath=`.status.ip`
|
||||
// +kubebuilder:printcolumn:name="Healthy",type=string,JSONPath=`.status.conditions[?(@.type=="Healthy")].status`
|
||||
// +kubebuilder:printcolumn:name="Age",type=date,JSONPath=`.metadata.creationTimestamp`
|
||||
|
||||
// Proxy is the Schema for the proxies API
|
||||
type Proxy struct {
|
||||
|
||||
@@ -25,6 +25,76 @@ import (
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
)
|
||||
|
||||
// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil.
|
||||
func (in *CloudInitSpec) DeepCopyInto(out *CloudInitSpec) {
|
||||
*out = *in
|
||||
if in.SecretRef != nil {
|
||||
in, out := &in.SecretRef, &out.SecretRef
|
||||
*out = new(SecretKeySelector)
|
||||
**out = **in
|
||||
}
|
||||
}
|
||||
|
||||
// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new CloudInitSpec.
|
||||
func (in *CloudInitSpec) DeepCopy() *CloudInitSpec {
|
||||
if in == nil {
|
||||
return nil
|
||||
}
|
||||
out := new(CloudInitSpec)
|
||||
in.DeepCopyInto(out)
|
||||
return out
|
||||
}
|
||||
|
||||
// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil.
|
||||
func (in *EndpointSpec) DeepCopyInto(out *EndpointSpec) {
|
||||
*out = *in
|
||||
}
|
||||
|
||||
// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new EndpointSpec.
|
||||
func (in *EndpointSpec) DeepCopy() *EndpointSpec {
|
||||
if in == nil {
|
||||
return nil
|
||||
}
|
||||
out := new(EndpointSpec)
|
||||
in.DeepCopyInto(out)
|
||||
return out
|
||||
}
|
||||
|
||||
// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil.
|
||||
func (in *HealthCheckSpec) DeepCopyInto(out *HealthCheckSpec) {
|
||||
*out = *in
|
||||
if in.ExpectedStatusCodes != nil {
|
||||
in, out := &in.ExpectedStatusCodes, &out.ExpectedStatusCodes
|
||||
*out = make([]int32, len(*in))
|
||||
copy(*out, *in)
|
||||
}
|
||||
}
|
||||
|
||||
// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new HealthCheckSpec.
|
||||
func (in *HealthCheckSpec) DeepCopy() *HealthCheckSpec {
|
||||
if in == nil {
|
||||
return nil
|
||||
}
|
||||
out := new(HealthCheckSpec)
|
||||
in.DeepCopyInto(out)
|
||||
return out
|
||||
}
|
||||
|
||||
// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil.
|
||||
func (in *PlacementSpec) DeepCopyInto(out *PlacementSpec) {
|
||||
*out = *in
|
||||
}
|
||||
|
||||
// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new PlacementSpec.
|
||||
func (in *PlacementSpec) DeepCopy() *PlacementSpec {
|
||||
if in == nil {
|
||||
return nil
|
||||
}
|
||||
out := new(PlacementSpec)
|
||||
in.DeepCopyInto(out)
|
||||
return out
|
||||
}
|
||||
|
||||
// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil.
|
||||
func (in *Proxy) DeepCopyInto(out *Proxy) {
|
||||
*out = *in
|
||||
@@ -87,9 +157,36 @@ func (in *ProxyList) DeepCopyObject() runtime.Object {
|
||||
// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil.
|
||||
func (in *ProxySpec) DeepCopyInto(out *ProxySpec) {
|
||||
*out = *in
|
||||
if in.Foo != nil {
|
||||
in, out := &in.Foo, &out.Foo
|
||||
*out = new(string)
|
||||
if in.Placement != nil {
|
||||
in, out := &in.Placement, &out.Placement
|
||||
*out = new(PlacementSpec)
|
||||
**out = **in
|
||||
}
|
||||
if in.CloudInit != nil {
|
||||
in, out := &in.CloudInit, &out.CloudInit
|
||||
*out = new(CloudInitSpec)
|
||||
(*in).DeepCopyInto(*out)
|
||||
}
|
||||
if in.Endpoint != nil {
|
||||
in, out := &in.Endpoint, &out.Endpoint
|
||||
*out = new(EndpointSpec)
|
||||
**out = **in
|
||||
}
|
||||
if in.Attributes != nil {
|
||||
in, out := &in.Attributes, &out.Attributes
|
||||
*out = make(map[string]string, len(*in))
|
||||
for key, val := range *in {
|
||||
(*out)[key] = val
|
||||
}
|
||||
}
|
||||
if in.HealthCheck != nil {
|
||||
in, out := &in.HealthCheck, &out.HealthCheck
|
||||
*out = new(HealthCheckSpec)
|
||||
(*in).DeepCopyInto(*out)
|
||||
}
|
||||
if in.MaxLeases != nil {
|
||||
in, out := &in.MaxLeases, &out.MaxLeases
|
||||
*out = new(int32)
|
||||
**out = **in
|
||||
}
|
||||
}
|
||||
@@ -114,6 +211,10 @@ func (in *ProxyStatus) DeepCopyInto(out *ProxyStatus) {
|
||||
(*in)[i].DeepCopyInto(&(*out)[i])
|
||||
}
|
||||
}
|
||||
if in.LastHealthCheckTime != nil {
|
||||
in, out := &in.LastHealthCheckTime, &out.LastHealthCheckTime
|
||||
*out = (*in).DeepCopy()
|
||||
}
|
||||
}
|
||||
|
||||
// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new ProxyStatus.
|
||||
@@ -125,3 +226,18 @@ func (in *ProxyStatus) DeepCopy() *ProxyStatus {
|
||||
in.DeepCopyInto(out)
|
||||
return out
|
||||
}
|
||||
|
||||
// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil.
|
||||
func (in *SecretKeySelector) DeepCopyInto(out *SecretKeySelector) {
|
||||
*out = *in
|
||||
}
|
||||
|
||||
// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new SecretKeySelector.
|
||||
func (in *SecretKeySelector) DeepCopy() *SecretKeySelector {
|
||||
if in == nil {
|
||||
return nil
|
||||
}
|
||||
out := new(SecretKeySelector)
|
||||
in.DeepCopyInto(out)
|
||||
return out
|
||||
}
|
||||
|
||||
233
cmd/main.go
233
cmd/main.go
@@ -14,29 +14,51 @@ See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
*/
|
||||
|
||||
// Package main is the composition root: it loads the provider config (fail
|
||||
// fast), assembles the provider registry, and wires the reconciler, health
|
||||
// engine, lease store, discovery API, orphan GC, and metrics onto one
|
||||
// controller-runtime manager.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"flag"
|
||||
"fmt"
|
||||
"os"
|
||||
goruntime "runtime"
|
||||
"time"
|
||||
|
||||
// Import all Kubernetes client auth plugins (e.g. Azure, GCP, OIDC, etc.)
|
||||
// to ensure that exec-entrypoint and run can make use of them.
|
||||
_ "k8s.io/client-go/plugin/pkg/client/auth"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/labels"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
utilruntime "k8s.io/apimachinery/pkg/util/runtime"
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/cache"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/healthz"
|
||||
"sigs.k8s.io/controller-runtime/pkg/log/zap"
|
||||
ctrlmetrics "sigs.k8s.io/controller-runtime/pkg/metrics"
|
||||
"sigs.k8s.io/controller-runtime/pkg/metrics/filters"
|
||||
metricsserver "sigs.k8s.io/controller-runtime/pkg/metrics/server"
|
||||
"sigs.k8s.io/controller-runtime/pkg/webhook"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/controller"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/discovery"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/gc"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/health"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/lease"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/metrics"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider/gcp"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider/kubernetes"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider/registry"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/version"
|
||||
// +kubebuilder:scaffold:imports
|
||||
)
|
||||
|
||||
@@ -56,12 +78,22 @@ func init() {
|
||||
func main() {
|
||||
var metricsAddr string
|
||||
var metricsCertPath, metricsCertName, metricsCertKey string
|
||||
var webhookCertPath, webhookCertName, webhookCertKey string
|
||||
var enableLeaderElection bool
|
||||
var probeAddr string
|
||||
var secureMetrics bool
|
||||
var enableHTTP2 bool
|
||||
var tlsOpts []func(*tls.Config)
|
||||
|
||||
var providersConfig string
|
||||
var discoveryAddr string
|
||||
var proxyNamespace string
|
||||
var healthWorkers int
|
||||
var gcInterval, gcMinAge time.Duration
|
||||
var gcAllowNamespaced bool
|
||||
var leaseCooldown, maxLeaseTTL time.Duration
|
||||
var showVersion bool
|
||||
var gcpWireFullPayloads bool
|
||||
|
||||
flag.StringVar(&metricsAddr, "metrics-bind-address", "0", "The address the metrics endpoint binds to. "+
|
||||
"Use :8443 for HTTPS or :8080 for HTTP, or leave as 0 to disable the metrics service.")
|
||||
flag.StringVar(&probeAddr, "health-probe-bind-address", ":8081", "The address the probe endpoint binds to.")
|
||||
@@ -70,22 +102,79 @@ func main() {
|
||||
"Enabling this will ensure there is only one active controller manager.")
|
||||
flag.BoolVar(&secureMetrics, "metrics-secure", true,
|
||||
"If set, the metrics endpoint is served securely via HTTPS. Use --metrics-secure=false to use HTTP instead.")
|
||||
flag.StringVar(&webhookCertPath, "webhook-cert-path", "", "The directory that contains the webhook certificate.")
|
||||
flag.StringVar(&webhookCertName, "webhook-cert-name", "tls.crt", "The name of the webhook certificate file.")
|
||||
flag.StringVar(&webhookCertKey, "webhook-cert-key", "tls.key", "The name of the webhook key file.")
|
||||
flag.StringVar(&metricsCertPath, "metrics-cert-path", "",
|
||||
"The directory that contains the metrics server certificate.")
|
||||
flag.StringVar(&metricsCertName, "metrics-cert-name", "tls.crt", "The name of the metrics server certificate file.")
|
||||
flag.StringVar(&metricsCertKey, "metrics-cert-key", "tls.key", "The name of the metrics server key file.")
|
||||
flag.BoolVar(&enableHTTP2, "enable-http2", false,
|
||||
"If set, HTTP/2 will be enabled for the metrics and webhook servers")
|
||||
"If set, HTTP/2 will be enabled for the metrics server")
|
||||
|
||||
flag.StringVar(&providersConfig, "providers-config", "",
|
||||
"Path to the providers config YAML. Required.")
|
||||
flag.StringVar(&discoveryAddr, "discovery-addr", ":8090",
|
||||
"Listen address of the discovery/lease HTTP API.")
|
||||
flag.StringVar(&proxyNamespace, "proxy-namespace", "",
|
||||
"Restrict the manager's cache to one namespace. Empty watches all namespaces. "+
|
||||
"Restricting also disables orphan GC unless --gc-allow-namespaced is set.")
|
||||
flag.IntVar(&healthWorkers, "health-workers", 8,
|
||||
"Number of concurrent health-probe workers.")
|
||||
flag.DurationVar(&gcInterval, "gc-interval", 10*time.Minute,
|
||||
"Interval between orphan GC sweeps.")
|
||||
flag.DurationVar(&gcMinAge, "gc-min-age", 10*time.Minute,
|
||||
"Minimum instance age before orphan GC may delete it.")
|
||||
flag.BoolVar(&gcAllowNamespaced, "gc-allow-namespaced", false,
|
||||
"Allow orphan GC to run although the cache is namespace-restricted. Dangerous: proxies "+
|
||||
"outside the namespace count as orphans and their instances get deleted.")
|
||||
flag.DurationVar(&leaseCooldown, "lease-cooldown", 15*time.Minute,
|
||||
"How long a reported proxy/target pair is excluded from lease selection.")
|
||||
flag.DurationVar(&maxLeaseTTL, "max-lease-ttl", time.Hour,
|
||||
"Maximum lease TTL a client may request.")
|
||||
flag.BoolVar(&showVersion, "version", false,
|
||||
"Print the commit the binary was built from and exit.")
|
||||
flag.BoolVar(&gcpWireFullPayloads, "gcp-wire-log-full-payloads", false,
|
||||
"Log GCP V(5) wire payloads verbatim instead of eliding fields larger than 1KiB.")
|
||||
|
||||
opts := zap.Options{
|
||||
Development: true,
|
||||
}
|
||||
opts.BindFlags(flag.CommandLine)
|
||||
flag.Parse()
|
||||
|
||||
if showVersion {
|
||||
fmt.Println(version.Resolve())
|
||||
os.Exit(0)
|
||||
}
|
||||
|
||||
ctrl.SetLogger(zap.New(zap.UseFlagOptions(&opts)))
|
||||
setupLog.Info("Starting egress-proxies-operator",
|
||||
"commit", version.Resolve(), "goVersion", goruntime.Version())
|
||||
ctx := ctrl.SetupSignalHandler()
|
||||
|
||||
// Providers load first and fail fast: a manager that comes up without
|
||||
// its backends would just convert every Proxy into an error loop.
|
||||
if providersConfig == "" {
|
||||
setupLog.Error(nil, "--providers-config is required")
|
||||
os.Exit(1)
|
||||
}
|
||||
cfg, err := provider.LoadConfigFile(providersConfig)
|
||||
if err != nil {
|
||||
setupLog.Error(err, "Failed to load providers config", "path", providersConfig)
|
||||
os.Exit(1)
|
||||
}
|
||||
providers, err := registry.Build(ctx, cfg, map[string]registry.Constructor{
|
||||
"kubernetes": kubernetes.New,
|
||||
"gcp": func(ctx context.Context, pc provider.ProviderConfig) (provider.Provider, error) {
|
||||
return gcp.NewWithWireOptions(ctx, pc, gcp.WireLogOptions{FullPayloads: gcpWireFullPayloads})
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
setupLog.Error(err, "Failed to build providers")
|
||||
os.Exit(1)
|
||||
}
|
||||
m := metrics.New()
|
||||
for name, p := range providers {
|
||||
providers[name] = provider.WithMetrics(name, p, m)
|
||||
}
|
||||
|
||||
// if the enable-http2 flag is false (the default), http/2 should be disabled
|
||||
// due to its vulnerabilities. More specifically, disabling http/2 will
|
||||
@@ -102,23 +191,6 @@ func main() {
|
||||
tlsOpts = append(tlsOpts, disableHTTP2)
|
||||
}
|
||||
|
||||
// Initial webhook TLS options
|
||||
webhookTLSOpts := tlsOpts
|
||||
webhookServerOptions := webhook.Options{
|
||||
TLSOpts: webhookTLSOpts,
|
||||
}
|
||||
|
||||
if len(webhookCertPath) > 0 {
|
||||
setupLog.Info("Initializing webhook certificate watcher using provided certificates",
|
||||
"webhook-cert-path", webhookCertPath, "webhook-cert-name", webhookCertName, "webhook-cert-key", webhookCertKey)
|
||||
|
||||
webhookServerOptions.CertDir = webhookCertPath
|
||||
webhookServerOptions.CertName = webhookCertName
|
||||
webhookServerOptions.KeyName = webhookCertKey
|
||||
}
|
||||
|
||||
webhookServer := webhook.NewServer(webhookServerOptions)
|
||||
|
||||
// Metrics endpoint is enabled in 'config/default/kustomization.yaml'. The Metrics options configure the server.
|
||||
// More info:
|
||||
// - https://pkg.go.dev/sigs.k8s.io/controller-runtime@v0.24.1/pkg/metrics/server
|
||||
@@ -139,12 +211,8 @@ func main() {
|
||||
|
||||
// If the certificate is not specified, controller-runtime will automatically
|
||||
// generate self-signed certificates for the metrics server. While convenient for development and testing,
|
||||
// this setup is not recommended for production.
|
||||
//
|
||||
// TODO(user): If you enable certManager, uncomment the following lines:
|
||||
// - [METRICS-WITH-CERTS] at config/default/kustomization.yaml to generate and use certificates
|
||||
// managed by cert-manager for the metrics server.
|
||||
// - [PROMETHEUS-WITH-CERTS] at config/prometheus/kustomization.yaml for TLS certification.
|
||||
// this setup is not recommended for production. This project doesn't use cert-manager (no admission
|
||||
// webhooks, no other consumer of managed certs) -- pass real certs via the flags below if needed.
|
||||
if len(metricsCertPath) > 0 {
|
||||
setupLog.Info("Initializing metrics certificate watcher using provided certificates",
|
||||
"metrics-cert-path", metricsCertPath, "metrics-cert-name", metricsCertName, "metrics-cert-key", metricsCertKey)
|
||||
@@ -154,33 +222,91 @@ func main() {
|
||||
metricsServerOptions.KeyName = metricsCertKey
|
||||
}
|
||||
|
||||
// The Secret cache is restricted to labelled cloud-init Secrets: the
|
||||
// operator has cluster-wide Secret read RBAC, and without the label
|
||||
// selector it would cache every Secret in scope.
|
||||
cacheOpts := cache.Options{
|
||||
ByObject: map[client.Object]cache.ByObject{
|
||||
&corev1.Secret{}: {
|
||||
Label: labels.SelectorFromSet(labels.Set{crawlv1alpha1.LabelCloudInit: "true"}),
|
||||
},
|
||||
},
|
||||
}
|
||||
if proxyNamespace != "" {
|
||||
cacheOpts.DefaultNamespaces = map[string]cache.Config{proxyNamespace: {}}
|
||||
}
|
||||
|
||||
mgr, err := ctrl.NewManager(ctrl.GetConfigOrDie(), ctrl.Options{
|
||||
Scheme: scheme,
|
||||
Metrics: metricsServerOptions,
|
||||
WebhookServer: webhookServer,
|
||||
HealthProbeBindAddress: probeAddr,
|
||||
Cache: cacheOpts,
|
||||
LeaderElection: enableLeaderElection,
|
||||
LeaderElectionID: "b47711d1.example.com",
|
||||
// LeaderElectionReleaseOnCancel defines if the leader should step down voluntarily
|
||||
// when the Manager ends. This requires the binary to immediately end when the
|
||||
// Manager is stopped, otherwise, this setting is unsafe. Setting this significantly
|
||||
// speeds up voluntary leader transitions as the new leader don't have to wait
|
||||
// LeaseDuration time first.
|
||||
//
|
||||
// In the default scaffold provided, the program ends immediately after
|
||||
// the manager stops, so would be fine to enable this option. However,
|
||||
// if you are doing or is intended to do any operation such as perform cleanups
|
||||
// after the manager stops then its usage might be unsafe.
|
||||
// LeaderElectionReleaseOnCancel: true,
|
||||
})
|
||||
if err != nil {
|
||||
setupLog.Error(err, "Failed to start manager")
|
||||
os.Exit(1)
|
||||
}
|
||||
|
||||
store := lease.NewStore(leaseCooldown)
|
||||
if err := mgr.Add(store); err != nil {
|
||||
setupLog.Error(err, "Failed to add lease store")
|
||||
os.Exit(1)
|
||||
}
|
||||
|
||||
engine := health.NewEngine(mgr.GetClient())
|
||||
engine.Workers = healthWorkers
|
||||
engine.Metrics = m
|
||||
if err := mgr.Add(engine); err != nil {
|
||||
setupLog.Error(err, "Failed to add health engine")
|
||||
os.Exit(1)
|
||||
}
|
||||
|
||||
if err := mgr.Add(&discovery.Server{
|
||||
Reader: mgr.GetClient(),
|
||||
Store: store,
|
||||
Addr: discoveryAddr,
|
||||
Token: os.Getenv("DISCOVERY_TOKEN"),
|
||||
MaxLeaseTTL: maxLeaseTTL,
|
||||
Metrics: m,
|
||||
}); err != nil {
|
||||
setupLog.Error(err, "Failed to add discovery server")
|
||||
os.Exit(1)
|
||||
}
|
||||
|
||||
if err := mgr.Add(&gc.Sweeper{
|
||||
Reader: mgr.GetClient(),
|
||||
Providers: providers,
|
||||
Interval: gcInterval,
|
||||
MinAge: gcMinAge,
|
||||
NamespaceRestricted: proxyNamespace != "",
|
||||
AllowNamespaced: gcAllowNamespaced,
|
||||
}); err != nil {
|
||||
setupLog.Error(err, "Failed to add orphan GC")
|
||||
os.Exit(1)
|
||||
}
|
||||
|
||||
if err := m.Register(ctrlmetrics.Registry,
|
||||
proxyPhaseCounts(mgr.GetClient()),
|
||||
func() int {
|
||||
total := 0
|
||||
for _, n := range store.Counts() {
|
||||
total += n
|
||||
}
|
||||
return total
|
||||
},
|
||||
); err != nil {
|
||||
setupLog.Error(err, "Failed to register metrics")
|
||||
os.Exit(1)
|
||||
}
|
||||
|
||||
if err := (&controller.ProxyReconciler{
|
||||
Client: mgr.GetClient(),
|
||||
Scheme: mgr.GetScheme(),
|
||||
Providers: providers,
|
||||
Health: engine,
|
||||
HealthEvents: engine.Events,
|
||||
}).SetupWithManager(mgr); err != nil {
|
||||
setupLog.Error(err, "Failed to create controller", "controller", "proxy")
|
||||
os.Exit(1)
|
||||
@@ -197,8 +323,31 @@ func main() {
|
||||
}
|
||||
|
||||
setupLog.Info("Starting manager")
|
||||
if err := mgr.Start(ctrl.SetupSignalHandler()); err != nil {
|
||||
if err := mgr.Start(ctx); err != nil {
|
||||
setupLog.Error(err, "Failed to run manager")
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
// proxyPhaseCounts reads phase counts from the cache at scrape time. Before
|
||||
// the cache has synced (or on any list error) it reports nothing rather
|
||||
// than something wrong.
|
||||
func proxyPhaseCounts(c client.Reader) func() map[string]int {
|
||||
return func() map[string]int {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer cancel()
|
||||
var list crawlv1alpha1.ProxyList
|
||||
if err := c.List(ctx, &list); err != nil {
|
||||
return nil
|
||||
}
|
||||
counts := map[string]int{}
|
||||
for i := range list.Items {
|
||||
phase := string(list.Items[i].Status.Phase)
|
||||
if phase == "" {
|
||||
phase = string(crawlv1alpha1.PhasePending)
|
||||
}
|
||||
counts[phase]++
|
||||
}
|
||||
return counts
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11,10 +11,31 @@ spec:
|
||||
kind: Proxy
|
||||
listKind: ProxyList
|
||||
plural: proxies
|
||||
shortNames:
|
||||
- px
|
||||
singular: proxy
|
||||
scope: Namespaced
|
||||
versions:
|
||||
- name: v1alpha1
|
||||
- additionalPrinterColumns:
|
||||
- jsonPath: .spec.mode
|
||||
name: Mode
|
||||
type: string
|
||||
- jsonPath: .spec.provider
|
||||
name: Provider
|
||||
type: string
|
||||
- jsonPath: .status.phase
|
||||
name: Phase
|
||||
type: string
|
||||
- jsonPath: .status.ip
|
||||
name: IP
|
||||
type: string
|
||||
- jsonPath: .status.conditions[?(@.type=="Healthy")].status
|
||||
name: Healthy
|
||||
type: string
|
||||
- jsonPath: .metadata.creationTimestamp
|
||||
name: Age
|
||||
type: date
|
||||
name: v1alpha1
|
||||
schema:
|
||||
openAPIV3Schema:
|
||||
description: Proxy is the Schema for the proxies API
|
||||
@@ -39,25 +60,198 @@ spec:
|
||||
spec:
|
||||
description: spec defines the desired state of Proxy
|
||||
properties:
|
||||
foo:
|
||||
description: foo is an example field of Proxy. Edit proxy_types.go
|
||||
to remove/update
|
||||
attributes:
|
||||
additionalProperties:
|
||||
type: string
|
||||
description: |-
|
||||
attributes are selection attributes exposed to the discovery API
|
||||
(geo, asn, purpose, ...). Deliberately separate from Kubernetes object
|
||||
labels, which stay an operator implementation concern.
|
||||
maxProperties: 32
|
||||
type: object
|
||||
cloudInit:
|
||||
description: |-
|
||||
cloudInit supplies the VM's cloud-init user-data. Only meaningful for
|
||||
Managed proxies. Changing the resolved content triggers replacement.
|
||||
properties:
|
||||
inline:
|
||||
description: inline is the literal cloud-init user-data.
|
||||
maxLength: 262144
|
||||
minLength: 1
|
||||
type: string
|
||||
secretRef:
|
||||
description: secretRef points at a Secret key holding the cloud-init
|
||||
user-data.
|
||||
properties:
|
||||
key:
|
||||
default: user-data
|
||||
description: key is the data key holding the cloud-init user-data.
|
||||
type: string
|
||||
name:
|
||||
description: name is the Secret's name.
|
||||
minLength: 1
|
||||
type: string
|
||||
required:
|
||||
- name
|
||||
type: object
|
||||
type: object
|
||||
x-kubernetes-validations:
|
||||
- message: exactly one of inline or secretRef must be set
|
||||
rule: has(self.inline) != has(self.secretRef)
|
||||
endpoint:
|
||||
description: |-
|
||||
endpoint identifies an External proxy's network location. Required
|
||||
iff mode is External; must be unset iff mode is Managed.
|
||||
properties:
|
||||
host:
|
||||
description: host is the proxy's hostname or IP address.
|
||||
minLength: 1
|
||||
type: string
|
||||
port:
|
||||
default: 3128
|
||||
description: port is the port the proxy listens on.
|
||||
format: int32
|
||||
maximum: 65535
|
||||
minimum: 1
|
||||
type: integer
|
||||
required:
|
||||
- host
|
||||
type: object
|
||||
healthCheck:
|
||||
default: {}
|
||||
description: |-
|
||||
healthCheck configures the through-the-proxy healthcheck. All nested
|
||||
fields are defaulted; the default={} marker ensures a proxy that omits
|
||||
healthCheck entirely still gets every nested default.
|
||||
properties:
|
||||
expectedStatusCodes:
|
||||
default:
|
||||
- 200
|
||||
- 204
|
||||
description: |-
|
||||
expectedStatusCodes are the HTTP status codes a probe response must
|
||||
match to count as successful.
|
||||
items:
|
||||
format: int32
|
||||
type: integer
|
||||
maxItems: 8
|
||||
type: array
|
||||
failureThreshold:
|
||||
default: 3
|
||||
description: |-
|
||||
failureThreshold is the number of consecutive failed probes required
|
||||
to transition Healthy -> False.
|
||||
format: int32
|
||||
minimum: 1
|
||||
type: integer
|
||||
intervalSeconds:
|
||||
default: 30
|
||||
description: intervalSeconds is the time between probes for a
|
||||
given proxy.
|
||||
format: int32
|
||||
minimum: 5
|
||||
type: integer
|
||||
probeURL:
|
||||
default: https://www.gstatic.com/generate_204
|
||||
description: probeURL is fetched through the proxy on every probe.
|
||||
minLength: 1
|
||||
type: string
|
||||
successThreshold:
|
||||
default: 1
|
||||
description: |-
|
||||
successThreshold is the number of consecutive successful probes
|
||||
required to transition Healthy -> True.
|
||||
format: int32
|
||||
minimum: 1
|
||||
type: integer
|
||||
timeoutSeconds:
|
||||
default: 5
|
||||
description: |-
|
||||
timeoutSeconds bounds a single probe, including the CONNECT tunnel
|
||||
setup and the TLS handshake through it.
|
||||
format: int32
|
||||
minimum: 1
|
||||
type: integer
|
||||
type: object
|
||||
maxLeases:
|
||||
default: 5
|
||||
description: |-
|
||||
maxLeases is the maximum number of concurrent leases handed out for
|
||||
this proxy. 0 means unleasable (list-only visibility). A pointer so an
|
||||
explicit 0 survives Go round-trips instead of being re-defaulted to 5.
|
||||
format: int32
|
||||
maximum: 1000
|
||||
minimum: 0
|
||||
type: integer
|
||||
mode:
|
||||
description: |-
|
||||
mode selects who owns the VM lifecycle: Managed (operator creates it)
|
||||
or External (operator only tracks and healthchecks it). Immutable.
|
||||
enum:
|
||||
- Managed
|
||||
- External
|
||||
type: string
|
||||
placement:
|
||||
description: |-
|
||||
placement is provider-opaque placement/size configuration. Only
|
||||
meaningful for Managed proxies.
|
||||
properties:
|
||||
image:
|
||||
description: image is the boot image reference.
|
||||
type: string
|
||||
machineType:
|
||||
description: machineType is the provider's machine/instance type
|
||||
(e.g. "e2-micro").
|
||||
type: string
|
||||
region:
|
||||
description: region is the provider's region identifier (e.g.
|
||||
"europe-west1").
|
||||
type: string
|
||||
zone:
|
||||
description: zone is the provider's zone identifier (e.g. "europe-west1-b").
|
||||
type: string
|
||||
type: object
|
||||
port:
|
||||
default: 3128
|
||||
description: |-
|
||||
port is the port the proxy listens on once the VM is up. Only
|
||||
meaningful for Managed proxies; External proxies use endpoint.port.
|
||||
format: int32
|
||||
maximum: 65535
|
||||
minimum: 1
|
||||
type: integer
|
||||
provider:
|
||||
description: |-
|
||||
provider is the name of a configured provider ("mock", "gcp-eu", ...).
|
||||
Required iff mode is Managed; must be unset iff mode is External.
|
||||
Immutable.
|
||||
maxLength: 63
|
||||
minLength: 1
|
||||
type: string
|
||||
required:
|
||||
- mode
|
||||
type: object
|
||||
x-kubernetes-validations:
|
||||
- message: mode is immutable
|
||||
rule: self.mode == oldSelf.mode
|
||||
- message: provider is immutable
|
||||
rule: has(self.provider) == has(oldSelf.provider) && (!has(self.provider)
|
||||
|| self.provider == oldSelf.provider)
|
||||
- message: provider is required when mode is Managed
|
||||
rule: self.mode != 'Managed' || has(self.provider)
|
||||
- message: provider must not be set when mode is External
|
||||
rule: self.mode != 'External' || !has(self.provider)
|
||||
- message: endpoint is required when mode is External
|
||||
rule: self.mode != 'External' || has(self.endpoint)
|
||||
- message: endpoint must not be set when mode is Managed
|
||||
rule: self.mode != 'Managed' || !has(self.endpoint)
|
||||
status:
|
||||
description: status defines the observed state of Proxy
|
||||
properties:
|
||||
conditions:
|
||||
description: |-
|
||||
conditions represent the current state of the Proxy resource.
|
||||
Each condition has a unique type and reflects the status of a specific aspect of the resource.
|
||||
|
||||
Standard condition types include:
|
||||
- "Available": the resource is fully functional
|
||||
- "Progressing": the resource is being created or updated
|
||||
- "Degraded": the resource failed to reach or maintain its desired state
|
||||
|
||||
The status of each condition is one of True, False, or Unknown.
|
||||
Standard types are Provisioned and Healthy.
|
||||
items:
|
||||
description: Condition contains details for one aspect of the current
|
||||
state of this API Resource.
|
||||
@@ -116,6 +310,37 @@ spec:
|
||||
x-kubernetes-list-map-keys:
|
||||
- type
|
||||
x-kubernetes-list-type: map
|
||||
ip:
|
||||
description: |-
|
||||
ip is the proxy's current IP address: the VM's address for Managed
|
||||
proxies, or spec.endpoint.host for External proxies.
|
||||
type: string
|
||||
lastHealthCheckTime:
|
||||
description: |-
|
||||
lastHealthCheckTime is the time of the last status-affecting probe,
|
||||
not the time of the most recent probe — probes that don't change the
|
||||
Healthy condition or move latency materially don't write status.
|
||||
format: date-time
|
||||
type: string
|
||||
latencyMillis:
|
||||
description: latencyMillis is the latency of the last status-affecting
|
||||
probe.
|
||||
format: int64
|
||||
type: integer
|
||||
observedGeneration:
|
||||
description: observedGeneration is the .metadata.generation last reconciled.
|
||||
format: int64
|
||||
type: integer
|
||||
phase:
|
||||
description: |-
|
||||
phase is a high-level, human-readable summary derived from conditions.
|
||||
One of Pending, Provisioning, Ready, Unhealthy, Deleting, Failed.
|
||||
type: string
|
||||
providerID:
|
||||
description: |-
|
||||
providerID is the opaque cloud resource ID returned by the provider.
|
||||
Empty for External proxies.
|
||||
type: string
|
||||
type: object
|
||||
required:
|
||||
- spec
|
||||
|
||||
@@ -4,13 +4,3 @@
|
||||
resources:
|
||||
- bases/crawl.example.com_proxies.yaml
|
||||
# +kubebuilder:scaffold:crdkustomizeresource
|
||||
|
||||
patches:
|
||||
# [WEBHOOK] To enable webhook, uncomment all the sections with [WEBHOOK] prefix.
|
||||
# patches here are for enabling the conversion webhook for each CRD
|
||||
# +kubebuilder:scaffold:crdkustomizewebhookpatch
|
||||
|
||||
# [WEBHOOK] To enable webhook, uncomment the following section
|
||||
# the following config is for teaching kustomize how to do kustomization for CRDs.
|
||||
#configurations:
|
||||
#- kustomizeconfig.yaml
|
||||
|
||||
@@ -1,12 +0,0 @@
|
||||
# This file is for teaching kustomize how to substitute name and namespace reference in CRD
|
||||
nameReference:
|
||||
- kind: Service
|
||||
version: v1
|
||||
fieldSpecs:
|
||||
- kind: CustomResourceDefinition
|
||||
version: v1
|
||||
group: apiextensions.k8s.io
|
||||
path: spec/conversion/webhook/clientConfig/service/name
|
||||
|
||||
varReference:
|
||||
- path: metadata/annotations
|
||||
18
config/default/discovery_service.yaml
Normal file
18
config/default/discovery_service.yaml
Normal file
@@ -0,0 +1,18 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
labels:
|
||||
control-plane: controller-manager
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
app.kubernetes.io/managed-by: kustomize
|
||||
name: controller-manager-discovery-service
|
||||
namespace: system
|
||||
spec:
|
||||
ports:
|
||||
- name: discovery
|
||||
port: 8090
|
||||
protocol: TCP
|
||||
targetPort: discovery
|
||||
selector:
|
||||
control-plane: controller-manager
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
@@ -18,20 +18,12 @@ resources:
|
||||
- ../crd
|
||||
- ../rbac
|
||||
- ../manager
|
||||
# [WEBHOOK] To enable webhook, uncomment all the sections with [WEBHOOK] prefix including the one in
|
||||
# crd/kustomization.yaml
|
||||
#- ../webhook
|
||||
# [CERTMANAGER] To enable cert-manager, uncomment all sections with 'CERTMANAGER'. 'WEBHOOK' components are required.
|
||||
#- ../certmanager
|
||||
# [PROMETHEUS] To enable prometheus monitor, uncomment all sections with 'PROMETHEUS'.
|
||||
#- ../prometheus
|
||||
# [METRICS] Expose the controller manager metrics service.
|
||||
- metrics_service.yaml
|
||||
# [NETWORK POLICY] Protect the /metrics endpoint and Webhook Server with NetworkPolicy.
|
||||
# Only Pod(s) running a namespace labeled with 'metrics: enabled' will be able to gather the metrics.
|
||||
# Only CR(s) which requires webhooks and are applied on namespaces labeled with 'webhooks: enabled' will
|
||||
# be able to communicate with the Webhook Server.
|
||||
#- ../network-policy
|
||||
# Expose the discovery/lease HTTP API inside the cluster.
|
||||
- discovery_service.yaml
|
||||
|
||||
# Uncomment the patches line if you enable Metrics
|
||||
patches:
|
||||
@@ -48,14 +40,9 @@ patches:
|
||||
# target:
|
||||
# kind: Deployment
|
||||
|
||||
# [WEBHOOK] To enable webhook, uncomment all the sections with [WEBHOOK] prefix including the one in
|
||||
# crd/kustomization.yaml
|
||||
#- path: manager_webhook_patch.yaml
|
||||
# target:
|
||||
# kind: Deployment
|
||||
|
||||
# [CERTMANAGER] To enable cert-manager, uncomment all sections with 'CERTMANAGER' prefix.
|
||||
# Uncomment the following replacements to add the cert-manager CA injection annotations
|
||||
# [METRICS-WITH-CERTS] Uncomment the following replacements together with the patch
|
||||
# above to wire the metrics Service name/namespace into the cert-manager Certificate
|
||||
# and the Prometheus ServiceMonitor TLS config.
|
||||
#replacements:
|
||||
# - source: # Uncomment the following block to enable certificates for metrics
|
||||
# kind: Service
|
||||
@@ -116,119 +103,3 @@ patches:
|
||||
# delimiter: '.'
|
||||
# index: 1
|
||||
# create: true
|
||||
|
||||
# - source: # Uncomment the following block if you have any webhook
|
||||
# kind: Service
|
||||
# version: v1
|
||||
# name: webhook-service
|
||||
# fieldPath: .metadata.name # Name of the service
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPaths:
|
||||
# - .spec.dnsNames.0
|
||||
# - .spec.dnsNames.1
|
||||
# options:
|
||||
# delimiter: '.'
|
||||
# index: 0
|
||||
# create: true
|
||||
# - source:
|
||||
# kind: Service
|
||||
# version: v1
|
||||
# name: webhook-service
|
||||
# fieldPath: .metadata.namespace # Namespace of the service
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPaths:
|
||||
# - .spec.dnsNames.0
|
||||
# - .spec.dnsNames.1
|
||||
# options:
|
||||
# delimiter: '.'
|
||||
# index: 1
|
||||
# create: true
|
||||
|
||||
# - source: # Uncomment the following block if you have a ValidatingWebhook (--programmatic-validation)
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert # This name should match the one in certificate.yaml
|
||||
# fieldPath: .metadata.namespace # Namespace of the certificate CR
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: ValidatingWebhookConfiguration
|
||||
# fieldPaths:
|
||||
# - .metadata.annotations.[cert-manager.io/inject-ca-from]
|
||||
# options:
|
||||
# delimiter: '/'
|
||||
# index: 0
|
||||
# create: true
|
||||
# - source:
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPath: .metadata.name
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: ValidatingWebhookConfiguration
|
||||
# fieldPaths:
|
||||
# - .metadata.annotations.[cert-manager.io/inject-ca-from]
|
||||
# options:
|
||||
# delimiter: '/'
|
||||
# index: 1
|
||||
# create: true
|
||||
|
||||
# - source: # Uncomment the following block if you have a DefaultingWebhook (--defaulting )
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPath: .metadata.namespace # Namespace of the certificate CR
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: MutatingWebhookConfiguration
|
||||
# fieldPaths:
|
||||
# - .metadata.annotations.[cert-manager.io/inject-ca-from]
|
||||
# options:
|
||||
# delimiter: '/'
|
||||
# index: 0
|
||||
# create: true
|
||||
# - source:
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPath: .metadata.name
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: MutatingWebhookConfiguration
|
||||
# fieldPaths:
|
||||
# - .metadata.annotations.[cert-manager.io/inject-ca-from]
|
||||
# options:
|
||||
# delimiter: '/'
|
||||
# index: 1
|
||||
# create: true
|
||||
|
||||
# - source: # Uncomment the following block if you have a ConversionWebhook (--conversion)
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPath: .metadata.namespace # Namespace of the certificate CR
|
||||
# targets: # Do not remove or uncomment the following scaffold marker; required to generate code for target CRD.
|
||||
# +kubebuilder:scaffold:crdkustomizecainjectionns
|
||||
# - source:
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPath: .metadata.name
|
||||
# targets: # Do not remove or uncomment the following scaffold marker; required to generate code for target CRD.
|
||||
# +kubebuilder:scaffold:crdkustomizecainjectionname
|
||||
|
||||
@@ -1,2 +1,9 @@
|
||||
resources:
|
||||
- manager.yaml
|
||||
- providers_config.yaml
|
||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||
kind: Kustomization
|
||||
images:
|
||||
- name: controller
|
||||
newName: example.com/egress-proxies-operator
|
||||
newTag: v0.0.1
|
||||
|
||||
@@ -63,12 +63,28 @@ spec:
|
||||
args:
|
||||
- --leader-elect
|
||||
- --health-probe-bind-address=:8081
|
||||
- --providers-config=/etc/proxy-operator/providers.yaml
|
||||
env:
|
||||
# Bearer token for the discovery API. Optional: without the
|
||||
# Secret the API serves unauthenticated (with a loud warning).
|
||||
# Create it with:
|
||||
# kubectl -n egress-proxies-operator-system create secret \
|
||||
# generic discovery-token --from-literal=token=<your-token>
|
||||
- name: DISCOVERY_TOKEN
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: discovery-token
|
||||
key: token
|
||||
optional: true
|
||||
image: controller:latest
|
||||
name: manager
|
||||
ports:
|
||||
- containerPort: 8081
|
||||
name: health
|
||||
protocol: TCP
|
||||
- containerPort: 8090
|
||||
name: discovery
|
||||
protocol: TCP
|
||||
securityContext:
|
||||
readOnlyRootFilesystem: true
|
||||
allowPrivilegeEscalation: false
|
||||
@@ -96,7 +112,13 @@ spec:
|
||||
requests:
|
||||
cpu: 10m
|
||||
memory: 64Mi
|
||||
volumeMounts: []
|
||||
volumes: []
|
||||
volumeMounts:
|
||||
- name: providers-config
|
||||
mountPath: /etc/proxy-operator
|
||||
readOnly: true
|
||||
volumes:
|
||||
- name: providers-config
|
||||
configMap:
|
||||
name: providers-config
|
||||
serviceAccountName: controller-manager
|
||||
terminationGracePeriodSeconds: 10
|
||||
|
||||
17
config/manager/providers_config.yaml
Normal file
17
config/manager/providers_config.yaml
Normal file
@@ -0,0 +1,17 @@
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: providers-config
|
||||
namespace: system
|
||||
labels:
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
app.kubernetes.io/managed-by: kustomize
|
||||
data:
|
||||
# Mounted at /etc/proxy-operator/providers.yaml (--providers-config).
|
||||
# The default ships only the kubernetes-pod provider so the operator runs
|
||||
# out of the box; add gcp entries (type: gcp, gcp.project: ...) for real
|
||||
# egress fleets — see config/samples/providers-config.yaml.
|
||||
providers.yaml: |
|
||||
providers:
|
||||
- name: kubernetes
|
||||
type: kubernetes
|
||||
@@ -1,27 +0,0 @@
|
||||
# This NetworkPolicy allows ingress traffic
|
||||
# with Pods running on namespaces labeled with 'metrics: enabled'. Only Pods on those
|
||||
# namespaces are able to gather data from the metrics endpoint.
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: NetworkPolicy
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
app.kubernetes.io/managed-by: kustomize
|
||||
name: allow-metrics-traffic
|
||||
namespace: system
|
||||
spec:
|
||||
podSelector:
|
||||
matchLabels:
|
||||
control-plane: controller-manager
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
policyTypes:
|
||||
- Ingress
|
||||
ingress:
|
||||
# This allows ingress traffic from any namespace with the label metrics: enabled
|
||||
- from:
|
||||
- namespaceSelector:
|
||||
matchLabels:
|
||||
metrics: enabled # Only from namespaces with this label
|
||||
ports:
|
||||
- port: 8443
|
||||
protocol: TCP
|
||||
@@ -1,2 +0,0 @@
|
||||
resources:
|
||||
- allow-metrics-traffic.yaml
|
||||
@@ -4,6 +4,24 @@ kind: ClusterRole
|
||||
metadata:
|
||||
name: manager-role
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- pods
|
||||
verbs:
|
||||
- create
|
||||
- delete
|
||||
- get
|
||||
- list
|
||||
- watch
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- secrets
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- watch
|
||||
- apiGroups:
|
||||
- crawl.example.com
|
||||
resources:
|
||||
|
||||
@@ -1,9 +0,0 @@
|
||||
apiVersion: crawl.example.com/v1alpha1
|
||||
kind: Proxy
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
app.kubernetes.io/managed-by: kustomize
|
||||
name: proxy-sample
|
||||
spec:
|
||||
# TODO(user): Add fields here
|
||||
@@ -1,4 +1,8 @@
|
||||
## Append samples of your project ##
|
||||
# providers-config.yaml is deliberately absent: it is a sample
|
||||
# --providers-config file, not a Kubernetes manifest.
|
||||
resources:
|
||||
- crawl_v1alpha1_proxy.yaml
|
||||
- proxy_kubernetes.yaml
|
||||
- proxy_gcp.yaml
|
||||
- proxy_external.yaml
|
||||
# +kubebuilder:scaffold:manifestskustomizesamples
|
||||
|
||||
21
config/samples/providers-config.yaml
Normal file
21
config/samples/providers-config.yaml
Normal file
@@ -0,0 +1,21 @@
|
||||
# Sample --providers-config file (not a Kubernetes manifest). In-cluster
|
||||
# this content lives in the providers-config ConfigMap
|
||||
# (config/manager/providers_config.yaml); for `make run-dev` a
|
||||
# kubernetes-only variant is at hack/providers-dev.yaml.
|
||||
#
|
||||
# Named provider instances: "gcp-eu" and "gcp-us" are two configs of the
|
||||
# same type. spec.provider on a Proxy refers to the name, not the type.
|
||||
providers:
|
||||
- name: kubernetes
|
||||
type: kubernetes
|
||||
# kubernetes:
|
||||
# image: ubuntu/squid:6.6-24.04_edge # the default
|
||||
- name: gcp-eu
|
||||
type: gcp
|
||||
gcp:
|
||||
project: my-project
|
||||
# network: default # VPC network name
|
||||
# networkTag: proxy-operator # firewall tag on created instances
|
||||
# diskSizeGb: 10
|
||||
# auth: Application Default Credentials (workload identity
|
||||
# in-cluster, gcloud ADC locally). No key-file plumbing.
|
||||
15
config/samples/proxy_external.yaml
Normal file
15
config/samples/proxy_external.yaml
Normal file
@@ -0,0 +1,15 @@
|
||||
# An External proxy: the VM exists outside the operator's control; the
|
||||
# operator only tracks and health-checks it through the endpoint. No
|
||||
# finalizer, no provider calls, and deleting the CR touches nothing.
|
||||
apiVersion: crawl.example.com/v1alpha1
|
||||
kind: Proxy
|
||||
metadata:
|
||||
name: proxy-external-sample
|
||||
spec:
|
||||
mode: External
|
||||
endpoint:
|
||||
host: 203.0.113.7
|
||||
port: 3128
|
||||
attributes:
|
||||
geo: eu
|
||||
purpose: crawl
|
||||
36
config/samples/proxy_gcp.yaml
Normal file
36
config/samples/proxy_gcp.yaml
Normal file
@@ -0,0 +1,36 @@
|
||||
# A Managed proxy backed by GCP: the operator creates a VM with an
|
||||
# ephemeral external IP and installs Squid via cloud-init. Requires a
|
||||
# providers-config entry named "gcp-eu" (see providers-config.yaml) and
|
||||
# Application Default Credentials with compute.instanceAdmin.v1.
|
||||
#
|
||||
# All three placement fields are required for GCP; the operator sets the
|
||||
# Proxy to Failed with a message naming any missing one.
|
||||
apiVersion: crawl.example.com/v1alpha1
|
||||
kind: Proxy
|
||||
metadata:
|
||||
name: proxy-gcp-sample
|
||||
spec:
|
||||
mode: Managed
|
||||
provider: gcp-eu
|
||||
placement:
|
||||
zone: europe-west1-b
|
||||
machineType: e2-micro
|
||||
image: projects/debian-cloud/global/images/family/debian-12
|
||||
port: 3128
|
||||
cloudInit:
|
||||
inline: |
|
||||
#cloud-config
|
||||
packages:
|
||||
- squid
|
||||
write_files:
|
||||
- path: /etc/squid/conf.d/proxy-operator.conf
|
||||
content: |
|
||||
http_port 3128
|
||||
http_access allow all
|
||||
via off
|
||||
forwarded_for off
|
||||
runcmd:
|
||||
- systemctl restart squid
|
||||
attributes:
|
||||
geo: eu
|
||||
purpose: crawl
|
||||
14
config/samples/proxy_kubernetes.yaml
Normal file
14
config/samples/proxy_kubernetes.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
# A Managed proxy backed by the kubernetes-pod provider: the operator runs
|
||||
# a real Squid pod in this cluster. This is the local-dev/CI sample — pods
|
||||
# share the cluster's egress IP, so it exercises the full lifecycle but
|
||||
# does not provide a distinct egress path (use the gcp provider for that).
|
||||
apiVersion: crawl.example.com/v1alpha1
|
||||
kind: Proxy
|
||||
metadata:
|
||||
name: proxy-kubernetes-sample
|
||||
spec:
|
||||
mode: Managed
|
||||
provider: kubernetes
|
||||
attributes:
|
||||
geo: local
|
||||
purpose: crawl
|
||||
323
docs/api.md
Normal file
323
docs/api.md
Normal file
@@ -0,0 +1,323 @@
|
||||
# Discovery API reference
|
||||
|
||||
The operator serves an HTTP API (the *discovery API*) that crawler clients
|
||||
use to find and lease egress proxies: list healthy proxies filtered by
|
||||
attributes, acquire a TTL-based lease on one, release it early, and report
|
||||
how a target site treated the proxy. It is implemented in
|
||||
[`internal/discovery`](../internal/discovery/) with lease state in
|
||||
[`internal/lease`](../internal/lease/); the only Kubernetes interaction is
|
||||
reading `Proxy` resources from the manager's cache.
|
||||
|
||||
## Base URL
|
||||
|
||||
The API listens on `:8090` (`--discovery-addr`) inside the manager pod and
|
||||
is exposed by a Service
|
||||
([config/default/discovery_service.yaml](../config/default/discovery_service.yaml)).
|
||||
|
||||
In-cluster:
|
||||
|
||||
```text
|
||||
http://egress-proxies-operator-controller-manager-discovery-service.egress-proxies-operator-system.svc.cluster.local:8090
|
||||
```
|
||||
|
||||
From a workstation, port-forward:
|
||||
|
||||
```sh
|
||||
kubectl -n egress-proxies-operator-system port-forward \
|
||||
svc/egress-proxies-operator-controller-manager-discovery-service 8090:8090 &
|
||||
```
|
||||
|
||||
Set `BASE_URL` to wherever you reach the API; all examples below use it:
|
||||
|
||||
```sh
|
||||
export BASE_URL=localhost:8090 # via the port-forward above
|
||||
# or, from inside the cluster:
|
||||
# export BASE_URL=http://egress-proxies-operator-controller-manager-discovery-service.egress-proxies-operator-system.svc.cluster.local:8090
|
||||
```
|
||||
|
||||
## Authentication
|
||||
|
||||
A single static bearer token, read from the `DISCOVERY_TOKEN` environment
|
||||
variable at startup. The shipped Deployment populates it from the
|
||||
`discovery-token` Secret (key `token`), which is **optional** — if the
|
||||
Secret is absent or the token is empty, the API serves **unauthenticated**
|
||||
(the manager logs a loud warning at startup). Create the Secret:
|
||||
|
||||
```sh
|
||||
kubectl -n egress-proxies-operator-system create secret generic discovery-token \
|
||||
--from-literal=token="$(openssl rand -hex 24)"
|
||||
```
|
||||
|
||||
Send the token on every request:
|
||||
|
||||
```sh
|
||||
export TOKEN=<the token>
|
||||
curl -s -H "Authorization: Bearer $TOKEN" "$BASE_URL/v1/proxies" | jq
|
||||
```
|
||||
|
||||
A missing or wrong token gets `401 {"error":"unauthorized",...}`.
|
||||
`GET /healthz` is always exempt.
|
||||
|
||||
The curl examples below omit the `-H "Authorization: Bearer $TOKEN"` flag
|
||||
for brevity — add it to every call when auth is enabled.
|
||||
|
||||
## Conventions
|
||||
|
||||
- Requests and responses are JSON. Errors share one envelope:
|
||||
|
||||
```json
|
||||
{"error": "<machine_code>", "message": "<human-readable text>"}
|
||||
```
|
||||
|
||||
- Request bodies are capped at **64 KiB** (larger bodies fail the JSON
|
||||
decode with `400 invalid_body`).
|
||||
- Proxies with a deletion timestamp (being finalized) are excluded from
|
||||
every response and never offered for lease.
|
||||
|
||||
## Configuration
|
||||
|
||||
| Setting | Default | Meaning |
|
||||
|---|---|---|
|
||||
| `--discovery-addr` | `:8090` | Listen address of the API |
|
||||
| `--max-lease-ttl` | `1h` | Maximum `ttlSeconds` a client may request |
|
||||
| `--lease-cooldown` | `15m` | Cooldown window applied on `rate_limited`/`banned` reports |
|
||||
| `DISCOVERY_TOKEN` (env) | empty | Bearer token; empty disables auth |
|
||||
|
||||
The shipped Deployment passes none of these flags, so the defaults apply.
|
||||
|
||||
## Endpoints
|
||||
|
||||
### `GET /healthz`
|
||||
|
||||
Liveness check. Unauthenticated, always `200` with body `ok`.
|
||||
|
||||
```sh
|
||||
curl -s "$BASE_URL/healthz"
|
||||
```
|
||||
|
||||
### `GET /v1/proxies` — list proxies
|
||||
|
||||
Query parameters (all optional):
|
||||
|
||||
| Parameter | Values | Effect |
|
||||
|---|---|---|
|
||||
| `healthy` | `true` \| `false` | Keep only proxies whose `Healthy` condition matches. Any other value → `400 invalid_query`. |
|
||||
| `attr.<key>` | any string | Exact match on `spec.attributes[<key>]`. Repeatable; **all** given pairs must match. |
|
||||
|
||||
List everything:
|
||||
|
||||
```sh
|
||||
curl -s "$BASE_URL/v1/proxies" | jq
|
||||
```
|
||||
|
||||
List healthy proxies in a given geo:
|
||||
|
||||
```sh
|
||||
curl -s "$BASE_URL/v1/proxies?healthy=true&attr.geo=eu" | jq
|
||||
```
|
||||
|
||||
Response — `200`, proxies sorted by `id`, an empty match is `200` with
|
||||
`"count": 0` (never `404`):
|
||||
|
||||
```json
|
||||
{
|
||||
"proxies": [
|
||||
{
|
||||
"id": "default/proxy-kubernetes-sample",
|
||||
"ip": "10.244.1.7",
|
||||
"port": 3128,
|
||||
"attributes": {"geo": "local"},
|
||||
"phase": "Ready",
|
||||
"healthy": true,
|
||||
"latencyMillis": 42,
|
||||
"activeLeases": 1,
|
||||
"maxLeases": 5
|
||||
}
|
||||
],
|
||||
"count": 1
|
||||
}
|
||||
```
|
||||
|
||||
Proxy object fields (the same shape appears inside lease responses):
|
||||
|
||||
| Field | Meaning |
|
||||
|---|---|
|
||||
| `id` | `namespace/name` of the `Proxy` resource; used as the stable key everywhere |
|
||||
| `ip` | Effective host — `spec.endpoint.host` for `External` proxies, `status.ip` for `Managed` (empty until the backing VM/pod is up) |
|
||||
| `port` | Effective port (default `3128`) |
|
||||
| `attributes` | `spec.attributes` — free-form selection labels (`geo`, `asn`, `purpose`, …); omitted when empty |
|
||||
| `phase` | `Pending` \| `Provisioning` \| `Ready` \| `Unhealthy` \| `Deleting` \| `Failed` |
|
||||
| `healthy` | `true` iff the `Healthy` condition is `True` (the through-the-proxy health probe passes) |
|
||||
| `latencyMillis` | Latency of the last status-affecting health probe |
|
||||
| `activeLeases` | Currently active leases on this proxy |
|
||||
| `maxLeases` | Lease capacity (default `5`; an explicit `0` means unleasable) |
|
||||
|
||||
### `POST /v1/leases` — acquire a lease
|
||||
|
||||
Picks a healthy proxy with free capacity matching the selector and grants
|
||||
an exclusive-slot, TTL-based lease on it.
|
||||
|
||||
Request body (every field optional; `{}` is valid):
|
||||
|
||||
```json
|
||||
{
|
||||
"selector": {"geo": "eu"},
|
||||
"ttlSeconds": 300,
|
||||
"target": "example.com"
|
||||
}
|
||||
```
|
||||
|
||||
| Field | Default | Meaning |
|
||||
|---|---|---|
|
||||
| `selector` | none | Attribute equality filter, same semantics as `attr.<key>` above |
|
||||
| `ttlSeconds` | `300` (5 min) | Lease lifetime; must be ≤ `--max-lease-ttl` (default 1 h), else `400 invalid_ttl` |
|
||||
| `target` | none | The site you intend to crawl; enables per-target cooldowns (see below) |
|
||||
|
||||
```sh
|
||||
curl -s -XPOST "$BASE_URL/v1/leases" \
|
||||
-d '{"selector":{"geo":"eu"},"ttlSeconds":300,"target":"example.com"}' | jq
|
||||
```
|
||||
|
||||
Success — `201`:
|
||||
|
||||
```json
|
||||
{
|
||||
"leaseID": "P3X6HHQTPCM5UTGVGE3B5UPS3A",
|
||||
"proxy": {
|
||||
"id": "default/proxy-eu-1",
|
||||
"ip": "34.88.10.20",
|
||||
"port": 3128,
|
||||
"attributes": {"geo": "eu"},
|
||||
"phase": "Ready",
|
||||
"healthy": true,
|
||||
"latencyMillis": 42,
|
||||
"activeLeases": 1,
|
||||
"maxLeases": 5
|
||||
},
|
||||
"expiresAt": "2026-08-11T22:05:00Z",
|
||||
"ttlSeconds": 300
|
||||
}
|
||||
```
|
||||
|
||||
Use `proxy.ip` and `proxy.port` as an HTTP proxy for the lease's lifetime:
|
||||
|
||||
```sh
|
||||
curl -x http://34.88.10.20:3128 https://example.com
|
||||
```
|
||||
|
||||
No match — `409` with diagnostic counts explaining why nothing qualified:
|
||||
|
||||
```json
|
||||
{
|
||||
"error": "no_match",
|
||||
"message": "no healthy proxy with free capacity matched the selector",
|
||||
"considered": 3,
|
||||
"atCapacity": 1,
|
||||
"inCooldown": 1,
|
||||
"unhealthy": 1
|
||||
}
|
||||
```
|
||||
|
||||
| Count | Meaning |
|
||||
|---|---|
|
||||
| `considered` | Proxies that matched the selector (before health/capacity checks) |
|
||||
| `atCapacity` | Skipped because `activeLeases >= maxLeases` |
|
||||
| `inCooldown` | Skipped because of an active cooldown for this target (or a global one) |
|
||||
| `unhealthy` | Skipped because the `Healthy` condition is not `True` |
|
||||
|
||||
Leases expire on their own — releasing is only needed to free the slot
|
||||
early. There is no renew/extend endpoint; acquire a new lease instead.
|
||||
|
||||
### `DELETE /v1/leases/{id}` — release early
|
||||
|
||||
Frees the lease's capacity slot immediately. Idempotent: always `204`,
|
||||
including for unknown or already-expired lease IDs.
|
||||
|
||||
```sh
|
||||
curl -si -XDELETE "$BASE_URL/v1/leases/P3X6HHQTPCM5UTGVGE3B5UPS3A"
|
||||
```
|
||||
|
||||
### `POST /v1/leases/{id}/report` — report an outcome
|
||||
|
||||
Tell the operator how the target site treated the proxy. This is the
|
||||
feedback signal that drives cooldowns.
|
||||
|
||||
Request body:
|
||||
|
||||
```json
|
||||
{"result": "rate_limited", "target": "example.com"}
|
||||
```
|
||||
|
||||
| Field | Values | Meaning |
|
||||
|---|---|---|
|
||||
| `result` | `ok` \| `rate_limited` \| `banned` | Anything else → `400 invalid_result` |
|
||||
| `target` | optional | Which site produced the result; falls back to the lease's `target`, then to global |
|
||||
|
||||
```sh
|
||||
curl -si -XPOST "$BASE_URL/v1/leases/P3X6HHQTPCM5UTGVGE3B5UPS3A/report" \
|
||||
-d '{"result":"rate_limited","target":"example.com"}'
|
||||
```
|
||||
|
||||
Responses: `204` on success, `404 unknown_lease` if the lease ID was never
|
||||
issued or has aged out.
|
||||
|
||||
Semantics:
|
||||
|
||||
- `ok` is a pure acknowledgement — nothing is recorded.
|
||||
- `rate_limited` and `banned` currently behave **identically**: both put
|
||||
the proxy in one cooldown window (default 15 min, `--lease-cooldown`)
|
||||
for the resolved target.
|
||||
- An expired lease remains reportable for one cooldown window past its
|
||||
TTL, so a late "we got rate-limited" still lands.
|
||||
|
||||
## Proxy selection and cooldowns
|
||||
|
||||
How `POST /v1/leases` picks among eligible proxies (healthy, matching the
|
||||
selector, not being deleted, not at capacity, not in cooldown), in order:
|
||||
|
||||
1. fewest `activeLeases` (least-loaded),
|
||||
2. lowest `latencyMillis`,
|
||||
3. lexicographic `id` (deterministic tie-break).
|
||||
|
||||
Cooldowns are keyed by **(proxy, target)**:
|
||||
|
||||
- A report **with a target** blocks that proxy only for lease requests
|
||||
naming the **same target**. Other targets — and requests with no
|
||||
target — still get the proxy.
|
||||
- A report **without a target**, on a lease that also had no target,
|
||||
creates a **global** cooldown: the proxy is blocked for *all* lease
|
||||
requests until the window passes. Always pass `target` on leases and
|
||||
reports unless you really mean "this proxy is bad for everyone".
|
||||
|
||||
## End-to-end example
|
||||
|
||||
```sh
|
||||
# 1. Acquire a lease for crawling example.com through an EU proxy
|
||||
LEASE=$(curl -s -XPOST "$BASE_URL/v1/leases" \
|
||||
-H "Authorization: Bearer $TOKEN" \
|
||||
-d '{"selector":{"geo":"eu"},"ttlSeconds":600,"target":"example.com"}')
|
||||
LEASE_ID=$(echo "$LEASE" | jq -r .leaseID)
|
||||
PROXY=$(echo "$LEASE" | jq -r '"\(.proxy.ip):\(.proxy.port)"')
|
||||
|
||||
# 2. Crawl through the leased proxy
|
||||
curl -x "http://$PROXY" https://example.com/some/page
|
||||
|
||||
# 3. Got a 429? Report it — example.com-bound leases will avoid this
|
||||
# proxy for the next 15 minutes
|
||||
curl -s -XPOST "$BASE_URL/v1/leases/$LEASE_ID/report" \
|
||||
-H "Authorization: Bearer $TOKEN" \
|
||||
-d '{"result":"rate_limited","target":"example.com"}'
|
||||
|
||||
# 4. Done early? Release the slot (otherwise the TTL frees it)
|
||||
curl -s -XDELETE "$BASE_URL/v1/leases/$LEASE_ID" \
|
||||
-H "Authorization: Bearer $TOKEN"
|
||||
```
|
||||
|
||||
## Caveats
|
||||
|
||||
- **Lease and cooldown state is in-memory and per-process.** An operator
|
||||
restart drops all active leases and cooldowns. Clients must tolerate a
|
||||
granted lease disappearing (a subsequent report returns `404`).
|
||||
- **Run a single replica.** The API is served by every manager replica but
|
||||
is not leader-elected, and lease state is not shared between replicas;
|
||||
the shipped Deployment pins `replicas: 1`.
|
||||
406
docs/architecture.md
Normal file
406
docs/architecture.md
Normal file
@@ -0,0 +1,406 @@
|
||||
# Architecture
|
||||
|
||||
## Components
|
||||
|
||||
| Component | Package | Runs as | Leader-elected | Role |
|
||||
|---|---|---|---|---|
|
||||
| Proxy CRD + helpers | `api/v1alpha1` | types | — | `Proxy` spec/status, CEL validation, defaulting, pure helpers |
|
||||
| Reconciler | `internal/controller` | controller | yes (with the manager) | the state machine: provision, replace, delete, represent health |
|
||||
| Provider contract | `internal/provider` | library | — | `Provider` interface, error taxonomy, deterministic naming, config, metrics decorator |
|
||||
| kubernetes provider | `internal/provider/kubernetes` | library | — | real Squid pods in this cluster (local dev/CI) |
|
||||
| gcp provider | `internal/provider/gcp` | library | — | Compute Engine VMs, four API calls, fire-and-forget ops |
|
||||
| Health engine | `internal/health` | Runnable | yes | through-the-proxy probes, thresholds, transition events |
|
||||
| Lease store | `internal/lease` | Runnable (expiry sweep) | no | in-memory leases + cooldowns, single mutex |
|
||||
| Discovery API | `internal/discovery` | Runnable | no | HTTP list/lease/release/report on `:8090` |
|
||||
| Orphan GC | `internal/gc` | Runnable | yes | deletes tagged instances whose CR is gone |
|
||||
| Metrics | `internal/metrics` | library | — | explicit registration, scrape-time collectors |
|
||||
| Composition root | `cmd/main.go` | binary | — | flags, provider registry, wires everything onto one manager |
|
||||
|
||||
## Event flow: cluster events → reconciler functions
|
||||
|
||||
Which functions run in response to which Kubernetes cluster events, as
|
||||
wired in `internal/controller/proxy_controller.go`.
|
||||
|
||||
### 1. How cluster events reach the reconciler
|
||||
|
||||
```text
|
||||
KUBERNETES CLUSTER EVENTS (wiring: SetupWithManager,
|
||||
───────────────────────── proxy_controller.go)
|
||||
|
||||
Proxy CR created / spec edited / Secret created / health transition
|
||||
status patched / delete requested updated / deleted (engine, see §6)
|
||||
│ │ │
|
||||
watch: For(&crawlv1alpha1.Proxy{}) watch: Watches( WatchesRawSource(
|
||||
│ &corev1.Secret{}, ...) source.Channel(
|
||||
│ │ r.HealthEvents, ...))
|
||||
│ r.proxiesForSecret(ctx, secret) │
|
||||
│ │ r.List(Proxies in namespace) │
|
||||
│ │ keeps those whose │
|
||||
│ │ spec.cloudInit.secretRef matches │
|
||||
│ ▼ │
|
||||
│ [reconcile.Request per matching Proxy] │
|
||||
▼ │ │
|
||||
┌────────────────────────────────────┴──────────────────────────┴──┐
|
||||
│ controller-runtime workqueue │◄── RequeueAfter
|
||||
│ (dedup by namespace/name, rate-limited, │ timers
|
||||
│ MaxConcurrentReconciles: 3) │◄── error backoff
|
||||
└──────────────────────────┬────────────────────────────────────────┘
|
||||
▼
|
||||
ProxyReconciler.Reconcile(ctx, req)
|
||||
```
|
||||
|
||||
The Secret watch makes *rotating a cloud-init Secret* a first-class event:
|
||||
it re-enqueues every Proxy referencing that Secret, which is how secret
|
||||
rotation triggers VM replacement even though the Proxy spec is untouched.
|
||||
|
||||
### 2. Inside `Reconcile` — dispatch and the single status write
|
||||
|
||||
```text
|
||||
Reconcile(ctx, req)
|
||||
│ r.Get(ctx, req.NamespacedName, &p) ── fetch the Proxy (NotFound → done)
|
||||
│ base := p.DeepCopy() ── snapshot for the diff
|
||||
│ defer patchStatusIfChanged(ctx, base, &p) ──────────────────────────┐
|
||||
│ │
|
||||
├─ p.DeletionTimestamp set ──► reconcileDelete(ctx, &p) │
|
||||
├─ p.Spec.Mode == External ──► reconcileExternal(ctx, &p) │
|
||||
└─ otherwise (Managed) ──────► reconcileManaged(ctx, &p) │
|
||||
▼
|
||||
patchStatusIfChanged (status.go)
|
||||
│ p.Status.ObservedGeneration = p.Generation
|
||||
│ p.Status.Phase = computePhase(&p)
|
||||
│ equality.Semantic.DeepEqual(base, p)?
|
||||
└─ changed → r.Status().Patch(...) ◄── the ONLY
|
||||
unchanged → no API call status write
|
||||
```
|
||||
|
||||
### 3. `reconcileManaged` — the state machine
|
||||
|
||||
```text
|
||||
reconcileManaged(ctx, p)
|
||||
│
|
||||
├─ controllerutil.AddFinalizer? ──► r.Update ──► return {} (watch event re-triggers)
|
||||
├─ permanent-failure latch (FindStatusCondition == PermanentError
|
||||
│ at this generation) ──► return {} (silent until spec edit)
|
||||
├─ r.Providers[p.Spec.Provider] missing ──► setProvisioned(PermanentError) → Failed
|
||||
│
|
||||
├─ resolveCloudInit(ctx, p) ──► r.Get(Secret) if secretRef (error → CloudInitError + backoff)
|
||||
├─ hash := specHash(p, cloudInit) (spechash.go)
|
||||
│
|
||||
├─ status.providerID == "" ─────────► prov.Create(CreateRequest{Name: NameFromUID(p.UID), ...})
|
||||
│ │ setSpecHash → r.Update (annotation)
|
||||
│ └ stage providerID + Provisioned=False/Provisioning
|
||||
│ ──► RequeueAfter: ProvisioningPoll
|
||||
│
|
||||
├─ annotation != hash, annotation == "" ──► adopt: setSpecHash → r.Update
|
||||
│ ──► RequeueAfter: RequeueNow
|
||||
├─ annotation != hash, annotation != "" ──► replaceInstance:
|
||||
│ prov.Get ─ NotFound → setSpecHash, clear ID/IP
|
||||
│ │ ──► RequeueNow (next pass creates)
|
||||
│ └ exists → prov.Delete, Provisioned=False/Replacing
|
||||
│ ──► RequeueAfter: DeletionPoll
|
||||
│
|
||||
└─ annotation == hash ──► prov.Get(providerID)
|
||||
├─ ErrNotFound ──► clear ID/IP ──► RequeueNow (next pass creates)
|
||||
├─ Provisioning ──► Provisioned=False ──► ProvisioningPoll
|
||||
├─ Running ──► status.ip = inst.IP,
|
||||
│ Provisioned=True/Created,
|
||||
│ applyHealth (see §6) ──► DriftPoll
|
||||
└─ Stopped/Termin. ──► prov.Delete (cattle) ──► DeletionPoll
|
||||
|
||||
any provider error ──► providerFailure(p, err) ── provider.Class(err):
|
||||
├─ ErrQuotaExceeded ──► condition QuotaExceeded ──► RequeueAfter: QuotaRetry (nil error)
|
||||
├─ ErrPermanent ──► condition PermanentError ──► phase Failed, no retry
|
||||
└─ ErrTransient ──► return err ──► workqueue exponential backoff
|
||||
```
|
||||
|
||||
### 4. `reconcileDelete` and `reconcileExternal`
|
||||
|
||||
```text
|
||||
reconcileDelete(ctx, p) reconcileExternal(ctx, p)
|
||||
├─ no finalizer ──► return {} │ status.ip = spec.endpoint.host
|
||||
├─ providerID == "" ──► RemoveFinalizer │ setProvisioned(True/ExternalEndpoint)
|
||||
│ → r.Update → object actually deleted │ applyHealth (see §6)
|
||||
├─ prov.Get → ErrNotFound ──► RemoveFinalizer └─ return {} (no finalizer,
|
||||
│ → r.Update → object actually deleted no provider calls ever)
|
||||
└─ exists ──► prov.Delete
|
||||
→ Provisioned=False/Deleting
|
||||
──► RequeueAfter: DeletionPoll (poll until gone)
|
||||
```
|
||||
|
||||
### 5. What provider calls do in the outside world
|
||||
|
||||
```text
|
||||
kubernetes pod provider (internal/provider/kubernetes/)
|
||||
prov.Create ──► buildPod (pure) ──► client.Create(corev1.Pod) ─┐ these cause Pod events,
|
||||
prov.Get ──► client.Get(Pod) → phase/IP → InstanceState │ but the operator does NOT
|
||||
prov.Delete ──► client.Delete(Pod, tolerate NotFound) │ watch Pods — it observes
|
||||
prov.ListByTag ─► client.List(Pods by labels, all namespaces) ─┘ them by polling prov.Get
|
||||
on each RequeueAfter tick
|
||||
|
||||
gcp provider (internal/provider/gcp/) — instances.{Insert,Get,Delete,AggregatedList}, nothing else
|
||||
prov.Create ──► buildInsertRequest (pure) ──► instances.Insert ─┐ fire-and-forget:
|
||||
409 alreadyExists = success (idempotent retry) │ Operation.Wait is never
|
||||
prov.Get ──► instances.Get → status/NatIP → InstanceState │ called; readiness is
|
||||
RUNNING without NatIP = still Provisioning │ discovered by Get polls,
|
||||
prov.Delete ──► instances.Delete (404 = success) │ exactly like the pod
|
||||
prov.ListByTag ─► AggregatedList(label filter, ─┘ provider
|
||||
ReturnPartialSuccess: true)
|
||||
providerID = zones/<zone>/instances/<name> — zone-qualified, so Get/Delete
|
||||
stay correct even mid-replacement after a zone edit
|
||||
```
|
||||
|
||||
The reconciler never watches provider-side resources (Pods or GCP VMs).
|
||||
All instance-state observation is poll-based through the `Provider`
|
||||
interface, so the same flow works identically for a cloud API that has no
|
||||
watch mechanism at all.
|
||||
|
||||
### 6. Health engine (`internal/health/`) — probes and transitions
|
||||
|
||||
The engine is a leader-elected manager Runnable with its own goroutines,
|
||||
independent of the workqueue. It owns health *state*; the reconciler owns
|
||||
its *representation* in status — that split keeps exactly one writer of
|
||||
`.status` and makes write-only-on-transition fall out for free.
|
||||
|
||||
```text
|
||||
Engine.Start(ctx) (engine.go)
|
||||
├─ spawns Workers (8) probe goroutines ◄─┐
|
||||
└─ ticker loop (Tick = 1s): │ jobs channel (non-blocking send;
|
||||
tick(ctx, now, jobs) │ saturated pool → retry next tick)
|
||||
│ Reader.List(Proxies) ── from the manager cache
|
||||
│ per proxy: skip if no IP/host or deleting (state pruned →
|
||||
│ a replaced instance starts with fresh counters)
|
||||
│ newState: seed verdict from an existing Healthy condition
|
||||
│ (leader handover), jitter first probe across the interval
|
||||
│ due && !inFlight ──► jobs ◄── probe worker picks up
|
||||
└ prune states for proxies gone from the cache
|
||||
│
|
||||
probe(ctx, proxyURL, hc, tls) (probe.go)
|
||||
│ fresh transport per probe, DisableKeepAlives=true
|
||||
│ (load-bearing: keep-alives would cache the CONNECT
|
||||
│ tunnel and later probes would never re-exercise it)
|
||||
│ https probe URL ⇒ CONNECT through the proxy + TLS inside
|
||||
└ success = err == nil AND expected status code
|
||||
│
|
||||
record(job, result, now) ── under one mutex
|
||||
│ counters: consecOK/consecFail; verdict flips only at
|
||||
│ successThreshold / failureThreshold
|
||||
│ emit ONLY on: first-ever verdict │ threshold flip │
|
||||
│ latency Δ > max(20ms, 50% of reported) rate-limited
|
||||
│ to one report per MinReportInterval (60s)
|
||||
▼
|
||||
Events chan (buffered 64, non-blocking send;
|
||||
on drop the reported markers do NOT advance → next probe retries)
|
||||
│
|
||||
▼
|
||||
source.Channel → workqueue → Reconcile (see §1)
|
||||
│
|
||||
▼
|
||||
r.applyHealth(p) ── reads Engine.Snapshot(key) (status.go)
|
||||
stages the Healthy condition + latencyMillis +
|
||||
lastHealthCheckTime; computePhase turns Provisioned=True
|
||||
+ Healthy=True/False into phase Ready / Unhealthy
|
||||
```
|
||||
|
||||
Consequence worth knowing: `status.lastHealthCheckTime` is the time of the
|
||||
last *status-affecting* probe, not the most recent probe — suppressed
|
||||
probes deliberately never write status. True probe recency will live in
|
||||
metrics (Step 9).
|
||||
|
||||
### 7. Discovery + lease API (`internal/discovery/`, `internal/lease/`)
|
||||
|
||||
HTTP-driven, not cluster-event-driven: crawler clients call in; the only
|
||||
Kubernetes interaction is reading Proxies from the manager's cache. The
|
||||
server is a non-leader-elected Runnable (all replicas would serve, but the
|
||||
deployment ships `replicas: 1` because lease state is per-process — an
|
||||
operator restart drops all leases and cooldowns, a documented caveat).
|
||||
Client-facing reference with request/response schemas and curl examples:
|
||||
[api.md](api.md).
|
||||
|
||||
```text
|
||||
crawler client
|
||||
│ Authorization: Bearer $DISCOVERY_TOKEN (empty token = auth disabled, loud startup warning)
|
||||
▼
|
||||
Server.handler() middleware, outermost first (server.go)
|
||||
recover → request-log → MaxBytesReader(64KiB) → bearer auth (constant-time; /healthz exempt)
|
||||
│
|
||||
├─ GET /healthz ──► 200 ok (unauthenticated)
|
||||
│
|
||||
├─ GET /v1/proxies?attr.k=v&healthy=true (handlers.go)
|
||||
│ Reader.List(Proxies) ── manager cache
|
||||
│ filter: attributes equality + Healthy condition
|
||||
│ + Store.Counts() for activeLeases
|
||||
│ ──► 200 {"proxies":[...], "count":N} (empty list is 200, not 404)
|
||||
│
|
||||
├─ POST /v1/leases {"selector":{...},"ttlSeconds":300,"target":"..."}
|
||||
│ Reader.List → filter selector; unhealthy matches counted, not offered
|
||||
│ Store.Acquire(healthy candidates, target, ttl) ── one lock: select+insert
|
||||
│ │ selection: fewest active leases, then latency, then name
|
||||
│ ├─ granted ──► 201 {leaseID, proxy:{...}, expiresAt, ttlSeconds}
|
||||
│ └─ ErrNoMatch ──► 409 {"error":"no_match", considered, atCapacity,
|
||||
│ inCooldown, unhealthy}
|
||||
│
|
||||
├─ DELETE /v1/leases/{id} ──► Store.Release ──► always 204 (idempotent)
|
||||
│
|
||||
└─ POST /v1/leases/{id}/report {"result":"ok|rate_limited|banned","target":"..."}
|
||||
Store.Report ── rate_limited/banned ⇒ cooldown[{proxy,target}] for
|
||||
│ CooldownWindow (target falls back: report → lease → global)
|
||||
├─ 204 │ 400 invalid_result │ 404 unknown_lease
|
||||
└─ an expired lease still resolves for CooldownWindow past its TTL —
|
||||
a late report lands exactly when the proxy is being rate-limited
|
||||
|
||||
Store.Start(ctx) ── manager Runnable, NOT leader-elected: sweeps expired
|
||||
leases + cooldowns; correctness never depends on the
|
||||
sweep (every read checks ExpiresAt against the clock)
|
||||
```
|
||||
|
||||
### 8. Orphan GC (`internal/gc/`) — the crash-safety net
|
||||
|
||||
Timer-driven, leader-elected (destructive ⇒ single writer). Exists for the
|
||||
one gap the reconciler cannot close alone: a crash after a provider Create
|
||||
but before the status write that records the instance.
|
||||
|
||||
```text
|
||||
Sweeper.Start(ctx) ── refuses to run when the cache is namespace-
|
||||
│ restricted unless --gc-allow-namespaced is explicit
|
||||
│ (an incomplete live set would "orphan" live VMs)
|
||||
└─ every Interval (10m; first sweep a full interval after start):
|
||||
sweep(ctx)
|
||||
│ Reader.List(Proxies) → live UID set
|
||||
│ List fails → skip the whole sweep (never guess)
|
||||
│ a CR with deletionTimestamp still counts as LIVE — its
|
||||
│ finalizer owns that deletion; GC racing it double-deletes
|
||||
└ per provider: ListByTag
|
||||
│ error → log, continue with the next provider
|
||||
└ delete only when ALL hold:
|
||||
has the proxy-operator-uid label (ownership proof)
|
||||
older than MinAge (10m) (not mid-create)
|
||||
UID matches no existing CR (truly orphaned)
|
||||
each kill logged loudly with provider, providerID, UID
|
||||
```
|
||||
|
||||
### 9. Metrics (`internal/metrics/`)
|
||||
|
||||
Registered explicitly from `cmd/main.go` (no `init()`; tests use fresh
|
||||
registries). Two kinds:
|
||||
|
||||
- **Scrape-time collectors** — `proxy_operator_proxies{phase}` and
|
||||
`proxy_operator_leases_active` read the cache / lease store at every
|
||||
scrape; reconcile-incremented gauges inevitably drift and leak series.
|
||||
- **Fed vectors** — `healthcheck_duration_seconds{proxy}` and
|
||||
`healthcheck_failures_total{proxy}` observe EVERY probe (status writes
|
||||
are transition-only; metrics carry the high-frequency signal), and the
|
||||
health engine deletes a proxy's series when it prunes its state;
|
||||
`lease_requests_total{outcome}` from the discovery handlers;
|
||||
`provider_requests_total{provider,op,result}` from the
|
||||
`provider.WithMetrics` decorator — the one place `Class()` is called
|
||||
purely for observability.
|
||||
|
||||
Each consuming package defines its own small recorder interface
|
||||
(`health.ProbeMetrics`, `discovery.LeaseMetrics`, `provider.RequestRecorder`);
|
||||
`metrics.Metrics` satisfies all of them structurally, so no package other
|
||||
than `cmd/main.go` imports the metrics package.
|
||||
|
||||
## Decisions
|
||||
|
||||
Judgment calls the spec left open, and deliberate deviations — recorded so
|
||||
they read as choices, not accidents. Chronological by build step.
|
||||
|
||||
- **Registry takes its constructor map as a parameter** instead of holding
|
||||
a package-level map: avoids the provider⇄registry import cycle and puts
|
||||
the wiring at the composition root, where it is visible.
|
||||
- **Mock provider replaced by the kubernetes-pod provider** (user
|
||||
decision, mid-build): a simulated in-memory provider was too far from
|
||||
the real system to build confidence in. Local dev/CI now runs real
|
||||
`ubuntu/squid` pods (Canonical's actively maintained image, verified
|
||||
50M+ pulls, pinned tag) in the operator's own cluster. Trade-off
|
||||
accepted: envtest has no kubelet, so end-to-end proof lives in the kind
|
||||
quickstart, and cluster pods share one egress IP — distinct egress
|
||||
paths remain the GCP provider's job.
|
||||
- **`RequeueAfter: RequeueNow` instead of the plan's `Requeue: true`:**
|
||||
`ctrl.Result.Requeue` is deprecated in controller-runtime v0.24; a fifth
|
||||
configurable interval (default 1s) keeps identical semantics and stays
|
||||
shrinkable in tests.
|
||||
- **Quota exhaustion is a wait, not a failure:** `ErrQuotaExceeded` sets a
|
||||
condition and requeues slowly (5m) with a nil error — off the backoff
|
||||
curve, out of the error log, and never `phase: Failed`. Only
|
||||
`ErrPermanent` latches Failed, keyed to the generation so a spec edit
|
||||
auto-recovers.
|
||||
- **The finalizer path never latches permanent failures:** a permanent
|
||||
error during deletion keeps retrying visibly instead — latching there
|
||||
would wedge the object forever with no path out but manual finalizer
|
||||
surgery.
|
||||
- **Health transitions travel reconciler-ward over a channel**
|
||||
(`source.Channel`), not direct status patches: `phase` derives from both
|
||||
provisioning and health, so two status writers would race and flap. One
|
||||
writer of status; the engine owns health *state*, the reconciler its
|
||||
*representation*; write-only-on-transition falls out for free.
|
||||
- **Health state seeds from the existing Healthy condition on leader
|
||||
handover** (verdict kept, counters zeroed, first probe jittered), so a
|
||||
healthy fleet doesn't flap to Unknown on restart — but a real
|
||||
transition still needs a full threshold run. A never-probed proxy skips
|
||||
the jitter and probes on the next tick: startup spread matters for
|
||||
restarts, not for a single new proxy.
|
||||
- **Latency suppression is `max(20ms, 50%)` + a 60s rate limit, and only
|
||||
while the verdict is healthy.** The spec's bare ">50% change" is
|
||||
undefined at 0 and lets a proxy jittering 40↔61ms write status forever;
|
||||
the healthy-only guard (found by test) stops a below-threshold success
|
||||
streak from emitting latency updates for a proxy still reported
|
||||
unhealthy. Consequence: `status.lastHealthCheckTime` means "last
|
||||
status-affecting probe" — true probe recency is in the metrics.
|
||||
- **Deterministic instance names** are `proxy-` + 16 chars of
|
||||
base32(SHA-256(CR UID)): legal for both GCP (`[a-z2-7]` ⊂ `[-a-z0-9]`,
|
||||
22 ≤ 63 chars) and Pod names, 80 bits against birthday collisions at a
|
||||
fleet of tens. The replacement VM therefore has the *same name* as the
|
||||
one being deleted — which is why replacement polls to NotFound before
|
||||
recreating instead of racing a 409.
|
||||
- **`banned` and `rate_limited` share one cooldown window:** a second
|
||||
duration knob the spec doesn't ask for; the report's semantic
|
||||
difference is preserved in the API but not the store.
|
||||
- **Report targets fall back report → lease → global**, so a client that
|
||||
leased with a target can't accidentally poison the proxy's global pool
|
||||
by omitting the target in its report.
|
||||
- **The 409 body's `considered` counts unhealthy matches too** (the store
|
||||
only ever sees healthy candidates): `considered = atCapacity +
|
||||
inCooldown + unhealthy + eligible-but-outranked`, keeping the numbers
|
||||
additive for a human debugging "why no proxy?".
|
||||
- **TTLs above `--max-lease-ttl` are a 400, not a silent clamp** — a
|
||||
client asking for a week should find out.
|
||||
- **Discovery is not leader-elected and ships `replicas: 1`:** caches
|
||||
start before non-leader-election runnables (verified in
|
||||
controller-runtime's ordering), and a leader-elected server would leave
|
||||
non-leader replicas as broken Service endpoints. One replica because
|
||||
lease state is per-process.
|
||||
- **GCP `Create` requires zone, machineType, and image** and fails
|
||||
`ErrPermanent` naming the missing field — inventing machine-type
|
||||
defaults would silently create billable VMs of arbitrary shape.
|
||||
- **Unknown GCP instance statuses map to `Stopped`:** the reconciler's
|
||||
answer to Stopped is delete-and-recreate, the always-safe move for
|
||||
cattle when the API grows a new state.
|
||||
- **Kubernetes 403s classify as `ErrPermanent`** even though quota
|
||||
exhaustion also surfaces as 403 (indistinguishable from RBAC denial in
|
||||
`apierrors`): not hammering an API server that may never allow the
|
||||
request is the safer default; a real ResourceQuota 403 forgoes the
|
||||
gentler quota backoff. Documented at the classification site.
|
||||
- **GC kills log at Info with a `WARNING:` prefix** — logr has no Warn
|
||||
level; the plan's "log at Warn" is met in spirit with provider,
|
||||
providerID, and UID always attached. Same convention as the
|
||||
discovery server's empty-token warning.
|
||||
- **GC trusts only provable orphans:** instances without the UID label
|
||||
are never deleted, a CR with a deletionTimestamp still counts as live
|
||||
(its finalizer owns that deletion), and an unreadable Proxy list skips
|
||||
the whole sweep. The namespace guard refuses to sweep a
|
||||
namespace-restricted cache without `--gc-allow-namespaced`.
|
||||
- **Cloud-init Secrets must carry `crawl.example.com/cloud-init: "true"`:**
|
||||
the manager caches only labelled Secrets (the operator holds
|
||||
cluster-wide Secret read RBAC — an unrestricted cache would hold every
|
||||
Secret in scope). Unlabelled referenced Secrets are invisible by
|
||||
construction, surfacing as `CloudInitError`.
|
||||
- **Events RBAC from the plan is omitted:** nothing wires an
|
||||
EventRecorder in the prototype, and granting verbs nothing uses would
|
||||
be RBAC lint noise. Add the marker together with the recorder if events
|
||||
land later.
|
||||
- **logr, not slog, inside controller paths:** the repo convention says
|
||||
`slog`, but `log.FromContext(ctx)` hands controller-runtime's logr
|
||||
logger to everything running under the manager — fighting that would
|
||||
mean two logging systems in one process. Noted as a deviation rather
|
||||
than silently ignored.
|
||||
94
docs/demo/create-gcp-proxies.sh
Executable file
94
docs/demo/create-gcp-proxies.sh
Executable file
@@ -0,0 +1,94 @@
|
||||
#!/bin/bash
|
||||
# Demo: create N Managed proxies backed by the gcp provider, named
|
||||
# proxy-gcp-demo-1 .. proxy-gcp-demo-N, each in a randomly picked EU zone
|
||||
# so the fleet gets egress IPs from different locations.
|
||||
#
|
||||
# Usage:
|
||||
# ./create-gcp-proxies.sh <count>
|
||||
#
|
||||
# Optional environment:
|
||||
# NAMESPACE namespace to create the proxies in (default: current context)
|
||||
# GCP_PROVIDER provider NAME from providers.yaml (default: gcp-eu)
|
||||
# ZONES space-separated zone list to pick from (default: EU zones below)
|
||||
set -euo pipefail
|
||||
|
||||
if [ $# -ne 1 ] || ! [[ "$1" =~ ^[1-9][0-9]*$ ]]; then
|
||||
echo "usage: $(basename "$0") <count> (positive integer)" >&2
|
||||
exit 1
|
||||
fi
|
||||
count="$1"
|
||||
provider="${GCP_PROVIDER:-gcp-eu}"
|
||||
|
||||
# GCP zones in the EU where e2-micro is generally available. Override with
|
||||
# ZONES="zone1 zone2 ..." if your project has quota only in some of them.
|
||||
default_zones=(
|
||||
europe-west1-b europe-west1-c europe-west1-d # Belgium
|
||||
europe-west2-a europe-west2-b europe-west2-c # London
|
||||
europe-west3-a europe-west3-b europe-west3-c # Frankfurt
|
||||
europe-west4-a europe-west4-b europe-west4-c # Netherlands
|
||||
europe-west6-a europe-west6-b europe-west6-c # Zurich
|
||||
europe-west8-a europe-west8-b europe-west8-c # Milan
|
||||
europe-west9-a europe-west9-b europe-west9-c # Paris
|
||||
europe-central2-a europe-central2-b europe-central2-c # Warsaw
|
||||
europe-north1-a europe-north1-b europe-north1-c # Finland
|
||||
europe-southwest1-a europe-southwest1-b europe-southwest1-c # Madrid
|
||||
)
|
||||
if [ -n "${ZONES:-}" ]; then
|
||||
read -r -a zones <<< "${ZONES}"
|
||||
else
|
||||
zones=("${default_zones[@]}")
|
||||
fi
|
||||
|
||||
if ! command -v kubectl &> /dev/null; then
|
||||
echo "ERROR: kubectl is required but not installed" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
ns_args=()
|
||||
if [ -n "${NAMESPACE:-}" ]; then
|
||||
ns_args=(-n "${NAMESPACE}")
|
||||
fi
|
||||
|
||||
for i in $(seq 1 "${count}"); do
|
||||
zone="${zones[RANDOM % ${#zones[@]}]}"
|
||||
echo "Creating proxy-gcp-demo-${i} in ${zone} ..."
|
||||
kubectl apply "${ns_args[@]}" -f - << EOF
|
||||
apiVersion: crawl.example.com/v1alpha1
|
||||
kind: Proxy
|
||||
metadata:
|
||||
name: proxy-gcp-demo-${i}
|
||||
spec:
|
||||
mode: Managed
|
||||
provider: ${provider} # must match a provider NAME in providers.yaml
|
||||
placement:
|
||||
zone: ${zone}
|
||||
machineType: e2-micro
|
||||
# debian-cloud images have no cloud-init, so spec.cloudInit (passed as
|
||||
# user-data metadata) would be silently ignored there. Ubuntu images do.
|
||||
image: projects/ubuntu-os-cloud/global/images/family/ubuntu-2404-lts-amd64
|
||||
port: 3128
|
||||
cloudInit:
|
||||
inline: |
|
||||
#cloud-config
|
||||
package_update: true
|
||||
packages:
|
||||
- squid
|
||||
write_files:
|
||||
- path: /etc/squid/conf.d/proxy-operator.conf
|
||||
content: |
|
||||
http_access allow all
|
||||
via off
|
||||
forwarded_for off
|
||||
runcmd:
|
||||
- systemctl restart squid
|
||||
attributes:
|
||||
geo: eu
|
||||
zone: ${zone}
|
||||
purpose: crawl
|
||||
EOF
|
||||
done
|
||||
|
||||
echo ""
|
||||
echo "Created ${count} proxies. VMs take a few minutes to provision and pass"
|
||||
echo "the health check. Watch them come up with:"
|
||||
echo " kubectl get px ${ns_args[*]:-} -w"
|
||||
48
docs/demo/create-kubernetes-proxies.sh
Executable file
48
docs/demo/create-kubernetes-proxies.sh
Executable file
@@ -0,0 +1,48 @@
|
||||
#!/bin/bash
|
||||
# Demo: create N Managed proxies backed by the kubernetes-pod provider,
|
||||
# named proxy-kubernetes-demo-1 .. proxy-kubernetes-demo-N. Pods share the
|
||||
# cluster's egress IP — this exercises the full lifecycle, not distinct
|
||||
# egress paths (use create-gcp-proxies.sh for that).
|
||||
#
|
||||
# Usage:
|
||||
# ./create-kubernetes-proxies.sh <count>
|
||||
#
|
||||
# Optional environment:
|
||||
# NAMESPACE namespace to create the proxies in (default: current context)
|
||||
set -euo pipefail
|
||||
|
||||
if [ $# -ne 1 ] || ! [[ "$1" =~ ^[1-9][0-9]*$ ]]; then
|
||||
echo "usage: $(basename "$0") <count> (positive integer)" >&2
|
||||
exit 1
|
||||
fi
|
||||
count="$1"
|
||||
|
||||
if ! command -v kubectl &> /dev/null; then
|
||||
echo "ERROR: kubectl is required but not installed" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
ns_args=()
|
||||
if [ -n "${NAMESPACE:-}" ]; then
|
||||
ns_args=(-n "${NAMESPACE}")
|
||||
fi
|
||||
|
||||
for i in $(seq 1 "${count}"); do
|
||||
echo "Creating proxy-kubernetes-demo-${i} ..."
|
||||
kubectl apply "${ns_args[@]}" -f - << EOF
|
||||
apiVersion: crawl.example.com/v1alpha1
|
||||
kind: Proxy
|
||||
metadata:
|
||||
name: proxy-kubernetes-demo-${i}
|
||||
spec:
|
||||
mode: Managed
|
||||
provider: kubernetes
|
||||
attributes:
|
||||
geo: local
|
||||
purpose: crawl
|
||||
EOF
|
||||
done
|
||||
|
||||
echo ""
|
||||
echo "Created ${count} proxies. Watch them come up with:"
|
||||
echo " kubectl get px ${ns_args[*]:-} -w"
|
||||
67
docs/demo/run-demo.sh
Executable file
67
docs/demo/run-demo.sh
Executable file
@@ -0,0 +1,67 @@
|
||||
#!/bin/bash
|
||||
# Demo driver: opens a tmux session with a 2x2 pane grid:
|
||||
#
|
||||
# top-left: show-egress-ips-table.sh in a 10s loop, run inside the
|
||||
# netshoot pod against the in-cluster discovery Service
|
||||
# top-right: watch -n3 kubectl get px
|
||||
# bottom-left: create-kubernetes-proxies.sh <count>
|
||||
# bottom-right: create-gcp-proxies.sh <count>
|
||||
#
|
||||
# Usage:
|
||||
# ./run-demo.sh
|
||||
#
|
||||
# Optional environment:
|
||||
# COUNT proxies each create script makes (default: 4)
|
||||
# SESSION tmux session name (default: proxy-demo; an existing
|
||||
# session with this name is killed and recreated)
|
||||
# NETSHOOT_POD pod to exec into for the egress-IP loop (default: netshoot)
|
||||
# DEMO_DIR where the demo scripts live (default: this script's dir)
|
||||
set -euo pipefail
|
||||
|
||||
COUNT="${COUNT:-4}"
|
||||
SESSION="${SESSION:-proxy-demo}"
|
||||
NETSHOOT_POD="${NETSHOOT_POD:-netshoot}"
|
||||
DEMO_DIR="${DEMO_DIR:-$(cd "$(dirname "$0")" && pwd)}"
|
||||
BASE_URL="http://egress-proxies-operator-controller-manager-discovery-service.egress-proxies-operator-system.svc.cluster.local:8090"
|
||||
|
||||
for tool in tmux kubectl; do
|
||||
if ! command -v "${tool}" &> /dev/null; then
|
||||
echo "ERROR: ${tool} is required but not installed" >&2
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
if ! kubectl get pod "${NETSHOOT_POD}" &> /dev/null; then
|
||||
echo "ERROR: pod ${NETSHOOT_POD} not found — start one with:" >&2
|
||||
echo " kubectl run netshoot --image=nicolaka/netshoot -- sleep infinity" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Copying show-egress-ips-table.sh into pod ${NETSHOOT_POD} ..."
|
||||
kubectl cp "${DEMO_DIR}/show-egress-ips-table.sh" "${NETSHOOT_POD}:/tmp/show-egress-ips-table.sh"
|
||||
|
||||
if tmux has-session -t "${SESSION}" 2> /dev/null; then
|
||||
echo "Killing existing tmux session ${SESSION}"
|
||||
tmux kill-session -t "${SESSION}"
|
||||
fi
|
||||
|
||||
# 2x2 grid: after these splits pane indexes are 0 top-left, 1 top-right,
|
||||
# 2 bottom-left, 3 bottom-right; tiled layout evens them into quarters.
|
||||
tmux new-session -d -s "${SESSION}"
|
||||
tmux split-window -h -t "${SESSION}:0"
|
||||
tmux split-window -v -t "${SESSION}:0.0"
|
||||
tmux split-window -v -t "${SESSION}:0.1"
|
||||
tmux select-layout -t "${SESSION}:0" tiled
|
||||
|
||||
loop_cmd="BASE_URL=${BASE_URL}; while true; do bash /tmp/show-egress-ips-table.sh \"\$BASE_URL\"; echo; sleep 10; done"
|
||||
tmux send-keys -t "${SESSION}:0.0" "kubectl exec -it ${NETSHOOT_POD} -- bash -c '${loop_cmd}'" C-m
|
||||
tmux send-keys -t "${SESSION}:0.1" "watch -n3 kubectl get px" C-m
|
||||
tmux send-keys -t "${SESSION}:0.2" "bash ${DEMO_DIR}/create-kubernetes-proxies.sh ${COUNT}" C-m
|
||||
tmux send-keys -t "${SESSION}:0.3" "bash ${DEMO_DIR}/create-gcp-proxies.sh ${COUNT}" C-m
|
||||
|
||||
tmux select-pane -t "${SESSION}:0.2"
|
||||
if [ -n "${TMUX:-}" ]; then
|
||||
tmux switch-client -t "${SESSION}"
|
||||
else
|
||||
tmux attach-session -t "${SESSION}"
|
||||
fi
|
||||
75
docs/demo/show-egress-ips-table.sh
Executable file
75
docs/demo/show-egress-ips-table.sh
Executable file
@@ -0,0 +1,75 @@
|
||||
#!/bin/bash
|
||||
# Demo: condensed-table variant of show-egress-ips.sh, made to fit a small
|
||||
# tmux pane. One line per proxy: which proxy the request goes through, its
|
||||
# endpoint, location (zone/geo attribute), and the egress IP the IP-echo
|
||||
# site saw — or unhealthy/FAILED.
|
||||
#
|
||||
# Usage:
|
||||
# ./show-egress-ips-table.sh <BASE_URL> e.g. ./show-egress-ips-table.sh localhost:8090
|
||||
#
|
||||
# Optional environment:
|
||||
# TOKEN bearer token for the discovery API (see docs/api.md)
|
||||
# IP_ECHO_URL site that returns the caller's IP as JSON with an "ip" field
|
||||
# (default: https://api.ipify.org?format=json)
|
||||
set -euo pipefail
|
||||
|
||||
if [ $# -ne 1 ]; then
|
||||
echo "usage: $(basename "$0") <BASE_URL> (e.g. localhost:8090)" >&2
|
||||
exit 1
|
||||
fi
|
||||
BASE_URL="$1"
|
||||
IP_ECHO_URL="${IP_ECHO_URL:-https://api.ipify.org?format=json}"
|
||||
|
||||
for tool in curl jq; do
|
||||
if ! command -v "${tool}" &> /dev/null; then
|
||||
echo "ERROR: ${tool} is required but not installed" >&2
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
auth_args=()
|
||||
if [ -n "${TOKEN:-}" ]; then
|
||||
auth_args=(-H "Authorization: Bearer ${TOKEN}")
|
||||
fi
|
||||
|
||||
if ! proxies_json=$(curl -sS --fail "${auth_args[@]}" "${BASE_URL}/v1/proxies") \
|
||||
|| ! echo "${proxies_json}" | jq -e . > /dev/null 2>&1; then
|
||||
echo "ERROR: could not fetch proxy list from ${BASE_URL}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
total=$(echo "${proxies_json}" | jq -r '.count')
|
||||
fmt="%-31s %-21s %-21s %s\n"
|
||||
|
||||
echo "${total} proxies @ $(date +%H:%M:%S)"
|
||||
# shellcheck disable=SC2059
|
||||
printf "${fmt}" "PROXY" "ENDPOINT" "LOCATION" "EGRESS-IP"
|
||||
|
||||
probed=0
|
||||
skipped=0
|
||||
failed=0
|
||||
while IFS= read -r proxy; do
|
||||
id=$(echo "${proxy}" | jq -r '.id')
|
||||
endpoint=$(echo "${proxy}" | jq -r '"\(.ip):\(.port)"')
|
||||
location=$(echo "${proxy}" | jq -r '.attributes.zone // .attributes.geo // "-"')
|
||||
healthy=$(echo "${proxy}" | jq -r '.healthy')
|
||||
|
||||
if [ "${healthy}" != "true" ]; then
|
||||
# shellcheck disable=SC2059
|
||||
printf "${fmt}" "${id}" "${endpoint}" "${location}" "(unhealthy)"
|
||||
skipped=$((skipped + 1))
|
||||
continue
|
||||
fi
|
||||
|
||||
if response=$(curl -sS --max-time 10 -x "http://${endpoint}" "${IP_ECHO_URL}" 2> /dev/null); then
|
||||
egress=$(echo "${response}" | jq -r '.ip // "?"' 2> /dev/null || echo "?")
|
||||
probed=$((probed + 1))
|
||||
else
|
||||
egress="FAILED"
|
||||
failed=$((failed + 1))
|
||||
fi
|
||||
# shellcheck disable=SC2059
|
||||
printf "${fmt}" "${id}" "${endpoint}" "${location}" "${egress}"
|
||||
done < <(echo "${proxies_json}" | jq -c '.proxies[]')
|
||||
|
||||
echo "-- ${probed} probed, ${skipped} unhealthy, ${failed} failed --"
|
||||
74
docs/demo/show-egress-ips.sh
Executable file
74
docs/demo/show-egress-ips.sh
Executable file
@@ -0,0 +1,74 @@
|
||||
#!/bin/bash
|
||||
# Demo: list proxies from the discovery API and show the egress IP each
|
||||
# healthy one provides, by calling an IP-echo site through it.
|
||||
#
|
||||
# Usage:
|
||||
# ./show-egress-ips.sh <BASE_URL> e.g. ./show-egress-ips.sh localhost:8090
|
||||
#
|
||||
# Optional environment:
|
||||
# TOKEN bearer token for the discovery API (see docs/api.md)
|
||||
# IP_ECHO_URL site that returns the caller's IP as JSON
|
||||
# (default: https://api.ipify.org?format=json)
|
||||
set -euo pipefail
|
||||
|
||||
if [ $# -ne 1 ]; then
|
||||
echo "usage: $(basename "$0") <BASE_URL> (e.g. localhost:8090)" >&2
|
||||
exit 1
|
||||
fi
|
||||
BASE_URL="$1"
|
||||
IP_ECHO_URL="${IP_ECHO_URL:-https://api.ipify.org?format=json}"
|
||||
|
||||
for tool in curl jq; do
|
||||
if ! command -v "${tool}" &> /dev/null; then
|
||||
echo "ERROR: ${tool} is required but not installed" >&2
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
|
||||
auth_args=()
|
||||
if [ -n "${TOKEN:-}" ]; then
|
||||
auth_args=(-H "Authorization: Bearer ${TOKEN}")
|
||||
fi
|
||||
|
||||
echo "Fetching proxies from ${BASE_URL}/v1/proxies ..."
|
||||
if ! proxies_json=$(curl -sS --fail "${auth_args[@]}" "${BASE_URL}/v1/proxies"); then
|
||||
echo "ERROR: could not fetch proxy list from ${BASE_URL}" >&2
|
||||
exit 1
|
||||
fi
|
||||
if ! echo "${proxies_json}" | jq -e . > /dev/null; then
|
||||
echo "ERROR: response from ${BASE_URL}/v1/proxies is not valid JSON" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
total=$(echo "${proxies_json}" | jq -r '.count')
|
||||
echo "Found ${total} proxies"
|
||||
echo ""
|
||||
|
||||
probed=0
|
||||
skipped=0
|
||||
failed=0
|
||||
while IFS= read -r proxy; do
|
||||
id=$(echo "${proxy}" | jq -r '.id')
|
||||
ip=$(echo "${proxy}" | jq -r '.ip')
|
||||
port=$(echo "${proxy}" | jq -r '.port')
|
||||
healthy=$(echo "${proxy}" | jq -r '.healthy')
|
||||
|
||||
if [ "${healthy}" != "true" ]; then
|
||||
echo "--- skipping ${id} (unhealthy) ---"
|
||||
echo ""
|
||||
skipped=$((skipped + 1))
|
||||
continue
|
||||
fi
|
||||
|
||||
echo "=== via ${id} — http://${ip}:${port} ==="
|
||||
if response=$(curl -sS --max-time 10 -x "http://${ip}:${port}" "${IP_ECHO_URL}"); then
|
||||
echo "${response}" | jq . 2> /dev/null || echo "${response}"
|
||||
probed=$((probed + 1))
|
||||
else
|
||||
echo "WARNING: request through ${id} failed" >&2
|
||||
failed=$((failed + 1))
|
||||
fi
|
||||
echo ""
|
||||
done < <(echo "${proxies_json}" | jq -c '.proxies[]')
|
||||
|
||||
echo "Done: ${total} proxies — ${probed} probed, ${skipped} skipped (unhealthy), ${failed} failed"
|
||||
255
docs/gcp-in-specific-project.md
Normal file
255
docs/gcp-in-specific-project.md
Normal file
@@ -0,0 +1,255 @@
|
||||
## Project egress-proxy
|
||||
|
||||
```bash
|
||||
PROJECT_ID=egress-proxy
|
||||
|
||||
# 1. Create the service account
|
||||
gcloud iam service-accounts create proxy-operator \
|
||||
--project ${PROJECT_ID} \
|
||||
--display-name "egress-proxies-operator"
|
||||
|
||||
# Output:
|
||||
# Created service account [proxy-operator].
|
||||
# Service account email: proxy-operator@egress-proxy.iam.gserviceaccount.com
|
||||
|
||||
# 2. Grant compute.instanceAdmin.v1 on the project
|
||||
gcloud projects add-iam-policy-binding ${PROJECT_ID} \
|
||||
--member "serviceAccount:proxy-operator@${PROJECT_ID}.iam.gserviceaccount.com" \
|
||||
--role roles/compute.instanceAdmin.v1
|
||||
|
||||
# Output:
|
||||
# ---------
|
||||
# Updated IAM policy for project [egress-proxy].
|
||||
# bindings:
|
||||
# - members:
|
||||
# - serviceAccount:proxy-operator@egress-proxy.iam.gserviceaccount.com
|
||||
# role: roles/compute.instanceAdmin.v1
|
||||
# - members:
|
||||
# - serviceAccount:541231138892@cloudservices.gserviceaccount.com
|
||||
# role: roles/compute.instanceGroupManagerServiceAgent
|
||||
# - members:
|
||||
# - serviceAccount:service-541231138892@compute-system.iam.gserviceaccount.com
|
||||
# role: roles/compute.serviceAgent
|
||||
# - members:
|
||||
# - user:admin@fujultimate.cz
|
||||
# role: roles/owner
|
||||
# etag: BwZYuFXko24=
|
||||
# version: 1
|
||||
|
||||
# 3. Create the JSON key (this is what goes into the Secret)
|
||||
SA_KEY_PATH=sa_key.json
|
||||
gcloud iam service-accounts keys create $SA_KEY_PATH \
|
||||
--iam-account proxy-operator@${PROJECT_ID}.iam.gserviceaccount.com
|
||||
|
||||
# output:
|
||||
# created key [fdff85174a8e80bbd684e76c4d9fe28e2f4b2ddf] of type [json] as [sa_key.json] for [proxy-operator@egress-proxy.iam.gserviceaccount.com]
|
||||
|
||||
# 4. A **firewall rule**: created VMs get network tag `proxy-operator` (the
|
||||
# default; configurable as `gcp.networkTag`), an ephemeral external IP,
|
||||
# and Squid listening on 3128.
|
||||
|
||||
gcloud compute firewall-rules create allow-proxy-operator \
|
||||
--project $PROJECT_ID \
|
||||
--network default \
|
||||
--allow tcp:3128 \
|
||||
--target-tags proxy-operator \
|
||||
--source-ranges 94.230.145.216/32
|
||||
```
|
||||
|
||||
## Phase 2 - resources in kube
|
||||
|
||||
```bash
|
||||
SA_KEY_PATH=sa_key.json
|
||||
kubectl -n egress-proxies-operator-system create secret generic gcp-credentials \
|
||||
--from-file=key.json=$SA_KEY_PATH
|
||||
```
|
||||
|
||||
|
||||
## Appendix - full manifests
|
||||
|
||||
```bash
|
||||
# crawl CR
|
||||
kubectl apply -f - <<'EOF'
|
||||
apiVersion: crawl.example.com/v1alpha1
|
||||
kind: Proxy
|
||||
metadata:
|
||||
name: proxy-gcp-sample
|
||||
spec:
|
||||
mode: Managed
|
||||
provider: gcp-eu # must match a provider NAME in providers.yaml
|
||||
placement:
|
||||
zone: europe-west1-b
|
||||
machineType: e2-micro
|
||||
# debian-cloud images have no cloud-init, so spec.cloudInit (passed as
|
||||
# user-data metadata) would be silently ignored there. Ubuntu images do.
|
||||
image: projects/ubuntu-os-cloud/global/images/family/ubuntu-2404-lts-amd64
|
||||
port: 3128
|
||||
cloudInit:
|
||||
inline: |
|
||||
#cloud-config
|
||||
package_update: true
|
||||
packages:
|
||||
- squid
|
||||
write_files:
|
||||
- path: /etc/squid/conf.d/proxy-operator.conf
|
||||
content: |
|
||||
http_access allow all
|
||||
via off
|
||||
forwarded_for off
|
||||
runcmd:
|
||||
- systemctl restart squid
|
||||
attributes:
|
||||
geo: eu
|
||||
purpose: crawl
|
||||
EOF
|
||||
|
||||
|
||||
|
||||
# configmap
|
||||
kubectl apply -f - <<'EOF'
|
||||
apiVersion: v1
|
||||
data:
|
||||
providers.yaml: |
|
||||
providers:
|
||||
- name: kubernetes
|
||||
type: kubernetes
|
||||
- name: gcp-eu # spec.provider on a Proxy refers to this NAME, not the type
|
||||
type: gcp
|
||||
gcp:
|
||||
project: egress-proxy
|
||||
# network: default # these three default as shown
|
||||
# networkTag: proxy-operator
|
||||
# diskSizeGb: 10
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/managed-by: kustomize
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
name: egress-proxies-operator-providers-config
|
||||
namespace: egress-proxies-operator-system
|
||||
EOF
|
||||
|
||||
# operator deployment
|
||||
kubectl apply -f - <<'EOF'
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
annotations:
|
||||
deployment.kubernetes.io/revision: "2"
|
||||
labels:
|
||||
app.kubernetes.io/managed-by: kustomize
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
control-plane: controller-manager
|
||||
name: egress-proxies-operator-controller-manager
|
||||
namespace: egress-proxies-operator-system
|
||||
spec:
|
||||
progressDeadlineSeconds: 600
|
||||
replicas: 1
|
||||
revisionHistoryLimit: 10
|
||||
selector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
control-plane: controller-manager
|
||||
strategy:
|
||||
rollingUpdate:
|
||||
maxSurge: 25%
|
||||
maxUnavailable: 25%
|
||||
type: RollingUpdate
|
||||
template:
|
||||
metadata:
|
||||
annotations:
|
||||
kubectl.kubernetes.io/default-container: manager
|
||||
labels:
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
control-plane: controller-manager
|
||||
spec:
|
||||
containers:
|
||||
- args:
|
||||
- --metrics-bind-address=:8443
|
||||
- --leader-elect
|
||||
- --health-probe-bind-address=:8081
|
||||
- --providers-config=/etc/proxy-operator/providers.yaml
|
||||
command:
|
||||
- /manager
|
||||
env:
|
||||
- name: DISCOVERY_TOKEN
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
key: token
|
||||
name: discovery-token
|
||||
optional: true
|
||||
- name: GOOGLE_APPLICATION_CREDENTIALS
|
||||
value: /var/secrets/gcp/key.json
|
||||
image: egress-proxies-operator:dev
|
||||
imagePullPolicy: IfNotPresent
|
||||
livenessProbe:
|
||||
failureThreshold: 3
|
||||
httpGet:
|
||||
path: /healthz
|
||||
port: 8081
|
||||
scheme: HTTP
|
||||
initialDelaySeconds: 15
|
||||
periodSeconds: 20
|
||||
successThreshold: 1
|
||||
timeoutSeconds: 1
|
||||
name: manager
|
||||
ports:
|
||||
- containerPort: 8081
|
||||
name: health
|
||||
protocol: TCP
|
||||
- containerPort: 8090
|
||||
name: discovery
|
||||
protocol: TCP
|
||||
readinessProbe:
|
||||
failureThreshold: 3
|
||||
httpGet:
|
||||
path: /readyz
|
||||
port: 8081
|
||||
scheme: HTTP
|
||||
initialDelaySeconds: 5
|
||||
periodSeconds: 10
|
||||
successThreshold: 1
|
||||
timeoutSeconds: 1
|
||||
resources:
|
||||
limits:
|
||||
cpu: 500m
|
||||
memory: 128Mi
|
||||
requests:
|
||||
cpu: 10m
|
||||
memory: 64Mi
|
||||
securityContext:
|
||||
allowPrivilegeEscalation: false
|
||||
capabilities:
|
||||
drop:
|
||||
- ALL
|
||||
readOnlyRootFilesystem: true
|
||||
terminationMessagePath: /dev/termination-log
|
||||
terminationMessagePolicy: File
|
||||
volumeMounts:
|
||||
- mountPath: /etc/proxy-operator
|
||||
name: providers-config
|
||||
readOnly: true
|
||||
- mountPath: /var/secrets/gcp
|
||||
name: gcp-credentials
|
||||
readOnly: true
|
||||
dnsPolicy: ClusterFirst
|
||||
restartPolicy: Always
|
||||
schedulerName: default-scheduler
|
||||
securityContext:
|
||||
runAsNonRoot: true
|
||||
seccompProfile:
|
||||
type: RuntimeDefault
|
||||
serviceAccount: egress-proxies-operator-controller-manager
|
||||
serviceAccountName: egress-proxies-operator-controller-manager
|
||||
terminationGracePeriodSeconds: 10
|
||||
volumes:
|
||||
- configMap:
|
||||
defaultMode: 420
|
||||
name: egress-proxies-operator-providers-config
|
||||
name: providers-config
|
||||
- name: gcp-credentials
|
||||
secret:
|
||||
defaultMode: 420
|
||||
secretName: gcp-credentials
|
||||
EOF
|
||||
```
|
||||
158
docs/gcp-vm-validation.md
Normal file
158
docs/gcp-vm-validation.md
Normal file
@@ -0,0 +1,158 @@
|
||||
# Validating real VM creation on GCP
|
||||
|
||||
Recipe for wiring the GCP provider into a live cluster and watching a
|
||||
`Proxy` CR create a real Compute Engine VM. Angle brackets mark values you
|
||||
supply: `<PROJECT_ID>`, `<SA_KEY_PATH>`, `<ZONE>`, `<CLUSTER_EGRESS_IP>`,
|
||||
`<REGISTRY_IMAGE>`.
|
||||
|
||||
The one important fact up front: **the operator takes no GCP credentials
|
||||
through its own config.** The client is built with Application Default
|
||||
Credentials (`internal/provider/gcp/gcp.go`, `New()`); there is no
|
||||
key-file field in the providers config. The only secret to prepare is a
|
||||
service-account JSON key, injected via the standard
|
||||
`GOOGLE_APPLICATION_CREDENTIALS` mechanism. On GKE you would use workload
|
||||
identity instead and skip the key entirely.
|
||||
|
||||
## 1. GCP-side prerequisites (prepared outside the cluster)
|
||||
|
||||
1. A project — `<PROJECT_ID>` — with the **Compute Engine API enabled**.
|
||||
2. A **service account** with `roles/compute.instanceAdmin.v1` on the
|
||||
project. The operator only calls instances
|
||||
`Insert`/`Get`/`Delete`/`AggregatedList` and does not attach a service
|
||||
account to the VMs it creates, so no `iam.serviceAccountUser` is
|
||||
needed.
|
||||
3. A **JSON key** for that service account, saved at `<SA_KEY_PATH>`.
|
||||
4. A **firewall rule**: created VMs get network tag `proxy-operator` (the
|
||||
default; configurable as `gcp.networkTag`), an ephemeral external IP,
|
||||
and Squid listening on 3128.
|
||||
|
||||
```sh
|
||||
gcloud compute firewall-rules create allow-proxy-operator \
|
||||
--project <PROJECT_ID> \
|
||||
--network default \
|
||||
--allow tcp:3128 \
|
||||
--target-tags proxy-operator \
|
||||
--source-ranges <CLUSTER_EGRESS_IP>/32
|
||||
```
|
||||
|
||||
The source range must cover the cluster's egress IP — the operator's
|
||||
CONNECT health probes originate there, and without the rule the Proxy
|
||||
hangs at `Running`/unhealthy instead of reaching `Ready`. ⚠️ The
|
||||
sample cloud-init configures `http_access allow all`, so on a public
|
||||
IP this is an open proxy — keep the source ranges tight.
|
||||
|
||||
## 2. Create the credentials Secret
|
||||
|
||||
Namespace is `egress-proxies-operator-system` after kustomize prefixing:
|
||||
|
||||
```sh
|
||||
kubectl -n egress-proxies-operator-system create secret generic gcp-credentials \
|
||||
--from-file=key.json=<SA_KEY_PATH>
|
||||
```
|
||||
|
||||
## 3. Add a GCP entry to the providers ConfigMap
|
||||
|
||||
Edit `config/manager/providers_config.yaml` (mounted at
|
||||
`/etc/proxy-operator/providers.yaml`):
|
||||
|
||||
```yaml
|
||||
providers:
|
||||
- name: kubernetes
|
||||
type: kubernetes
|
||||
- name: gcp-eu # spec.provider on a Proxy refers to this NAME, not the type
|
||||
type: gcp
|
||||
gcp:
|
||||
project: <PROJECT_ID>
|
||||
# network: default # these three default as shown
|
||||
# networkTag: proxy-operator
|
||||
# diskSizeGb: 10
|
||||
```
|
||||
|
||||
The config is validated fail-fast at startup — a typo shows up
|
||||
immediately in the manager log, not on first use.
|
||||
|
||||
## 4. Mount the Secret and point ADC at it
|
||||
|
||||
In `config/manager/manager.yaml`, add to the manager container:
|
||||
|
||||
```yaml
|
||||
env:
|
||||
- name: GOOGLE_APPLICATION_CREDENTIALS
|
||||
value: /var/secrets/gcp/key.json
|
||||
volumeMounts:
|
||||
- name: gcp-credentials
|
||||
mountPath: /var/secrets/gcp
|
||||
readOnly: true
|
||||
volumes:
|
||||
- name: gcp-credentials
|
||||
secret:
|
||||
secretName: gcp-credentials
|
||||
```
|
||||
|
||||
(`volumeMounts` merges into the existing container list; `volumes` into
|
||||
the existing pod-level list.)
|
||||
|
||||
## 5. Deploy and create the Proxy
|
||||
|
||||
```sh
|
||||
make deploy IMG=<REGISTRY_IMAGE>
|
||||
```
|
||||
|
||||
`config/samples/proxy_gcp.yaml` is usable as-is once `spec.provider`
|
||||
matches the name from step 3. All three placement fields are mandatory
|
||||
for GCP — a missing one sets the Proxy to `Failed` with a message naming
|
||||
it:
|
||||
|
||||
```yaml
|
||||
spec:
|
||||
mode: Managed
|
||||
provider: gcp-eu
|
||||
placement:
|
||||
zone: <ZONE> # e.g. europe-west1-b
|
||||
machineType: e2-micro
|
||||
image: projects/debian-cloud/global/images/family/debian-12
|
||||
```
|
||||
|
||||
```sh
|
||||
kubectl apply -f config/samples/proxy_gcp.yaml
|
||||
```
|
||||
|
||||
## 6. What you should see
|
||||
|
||||
```sh
|
||||
kubectl get proxy -w
|
||||
```
|
||||
|
||||
`Provisioning` → `Running` (VM's external IP published in status) →
|
||||
`Ready` (CONNECT health probe succeeded through the public IP). Then:
|
||||
|
||||
```sh
|
||||
# the VM exists and carries the GC labels
|
||||
gcloud compute instances list --project <PROJECT_ID> \
|
||||
--filter 'labels.proxy-operator-managed=yes'
|
||||
|
||||
# the proxy actually tunnels — should print the VM's external IP
|
||||
curl -x http://<EXTERNAL_IP>:3128 https://ifconfig.me
|
||||
```
|
||||
|
||||
Cleanup — the finalizer deletes the VM:
|
||||
|
||||
```sh
|
||||
kubectl delete proxy proxy-gcp-sample
|
||||
gcloud compute instances list --project <PROJECT_ID> # should be empty again
|
||||
```
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **The orphan GC sweeps the whole project**: any VM labeled
|
||||
`proxy-operator-managed=yes` whose UID does not match a live Proxy CR
|
||||
in *this* cluster is deleted once past the age threshold. Do not point
|
||||
two operator installs at the same project, and do not hand-create VMs
|
||||
with that label.
|
||||
- **VM creation is fire-and-forget** — the provider never waits on the
|
||||
insert operation; progress is discovered by polling `Get`. A quota
|
||||
error or bad image name surfaces on the Proxy's status/conditions a
|
||||
reconcile later, not synchronously. `kubectl describe proxy` is the
|
||||
place to look when something stalls.
|
||||
- **e2-micro costs pennies but is not free everywhere** — remember to
|
||||
delete the CR (or check `gcloud compute instances list`) when done.
|
||||
1131
docs/plans-executions/2026-08-07-1747-proxy-operator.md
Normal file
1131
docs/plans-executions/2026-08-07-1747-proxy-operator.md
Normal file
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,42 @@
|
||||
# Execution: Verbose (V-level) logging in the GCP provider
|
||||
|
||||
Plan: `docs/plans/2026-08-11-1742-gcp-provider-verbose-logging.md`
|
||||
|
||||
- [x] Step 0 — Save and commit the plan
|
||||
- [x] Step 1 — V(1)/V(2) logging in `internal/provider/gcp` (gcp.go, errors.go)
|
||||
- [x] Step 2 — Tests (verbosity tiers, error detail, cloud-init leak guard)
|
||||
- [ ] Step 3 — CHANGELOG entry (after the user confirms it works live)
|
||||
|
||||
## Step 1 — logging in the provider
|
||||
|
||||
Went as planned: context-carried logger (`logf.FromContext(ctx).WithName("gcp")`),
|
||||
V(1) one line per API call, V(2) request/list detail, opNames captured from the
|
||||
`instancesAPI` seam instead of being discarded. `logAPIError` lives in
|
||||
`errors.go` (next to `classify`, whose imports it shares) rather than `gcp.go`
|
||||
as loosely implied by the plan — same package, so no behavioural difference.
|
||||
These are the first `.V(n)` calls and the first logging import anywhere under
|
||||
`internal/provider/`.
|
||||
|
||||
Worth noting: the gopls `errorsastype` suggestion fired on the new
|
||||
`errors.As` in `logAPIError` (Go's newer `errors.AsType`); kept `errors.As`
|
||||
for consistency with the three existing uses in the same file. Same for the
|
||||
`newexpr` (`proto.String` → `new`) suggestions — the codebase consistently
|
||||
uses `proto.String`.
|
||||
|
||||
## Step 2 — tests
|
||||
|
||||
`funcr.New` as the capturing sink, injected via `logr.NewContext`, exactly the
|
||||
seam the plan predicted. One deviation: instead of a single
|
||||
`TestLogging_verbosity` table, it split into three tests — `_verbosityTiers`
|
||||
(table over V=0/1/2, incl. the cloud-init sentinel leak assertion),
|
||||
`_apiErrorKeepsHTTPDetail` (403 quotaExceeded keeps `httpStatus`/reason at
|
||||
V(1)), and `_treatedAsSuccessPathsAreExplicit` (409-on-create /
|
||||
404-on-delete each log their "treated as success" line) — the last two
|
||||
exercise fake error wiring that didn't fit the tier table cleanly.
|
||||
|
||||
Verified with:
|
||||
|
||||
```bash
|
||||
go test -race ./internal/provider/gcp/
|
||||
go test ./...
|
||||
```
|
||||
61
docs/plans-executions/2026-08-11-1802-bake-commit-version.md
Normal file
61
docs/plans-executions/2026-08-11-1802-bake-commit-version.md
Normal file
@@ -0,0 +1,61 @@
|
||||
# Execution: Bake the git commit into the operator binary and log it at startup
|
||||
|
||||
Plan: `docs/plans/2026-08-11-1802-bake-commit-version.md`
|
||||
|
||||
- [x] Step 0 — Save and commit the plan
|
||||
- [x] Step 1 — `internal/version` package + tests
|
||||
- [x] Step 2 — `cmd/main.go`: `--version` flag + startup log line
|
||||
- [x] Step 3 — Makefile `GIT_COMMIT` + `--build-arg` wiring
|
||||
- [x] Step 4 — Dockerfile `-ldflags` stamp + OCI revision label
|
||||
- [ ] Step 5 — CHANGELOG entry (after the user confirms it works live)
|
||||
|
||||
## Steps 1–4
|
||||
|
||||
Mostly as planned. One deviation worth recording: the plan claimed
|
||||
`build`/`run` targets need no changes because Go's automatic VCS stamp covers
|
||||
host builds — that turned out to be only half true. Go skips VCS stamping
|
||||
when the build target is a *file argument* rather than a package pattern, and
|
||||
the Makefile's `build` target used `go build -o bin/manager cmd/main.go`.
|
||||
Verified empirically:
|
||||
|
||||
```bash
|
||||
go build -o bin/manager cmd/main.go && ./bin/manager --version # unknown (no vcs settings)
|
||||
go build -o bin/manager ./cmd && ./bin/manager --version # e4d2a191d0c2-dirty
|
||||
```
|
||||
|
||||
So `build:` now uses `go build -o bin/manager ./cmd`. `go run` never stamps
|
||||
VCS info regardless of invocation form — `make run`/`run-dev` print
|
||||
`commit=unknown`, which is acceptable for dev loops (the Dockerfile path uses
|
||||
the explicit ldflags stamp and is unaffected; it kept `cmd/main.go`).
|
||||
|
||||
The ldflags path was verified independently:
|
||||
|
||||
```bash
|
||||
go build -ldflags "-X gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/version.Commit=deadbeef1234" \
|
||||
-o bin/manager-stamped cmd/main.go
|
||||
./bin/manager-stamped --version # deadbeef1234
|
||||
```
|
||||
|
||||
Worth noting: `cmd/main.go` imports k8s apimachinery as `runtime`, so the
|
||||
stdlib runtime needed an alias (`goruntime "runtime"`) for
|
||||
`goruntime.Version()` in the startup line. The `--version` check happens
|
||||
right after `flag.Parse()`, before logger and manager setup, so it works
|
||||
without a kubeconfig.
|
||||
|
||||
## Verification
|
||||
|
||||
```bash
|
||||
go vet ./... && go test ./... # all green, incl. new resolve() table tests
|
||||
go build -o bin/manager ./cmd && ./bin/manager --version # e4d2a191d0c2-dirty
|
||||
```
|
||||
|
||||
The image-level check initially failed (Docker daemon not running); after
|
||||
the daemon was started it passed in full:
|
||||
|
||||
```bash
|
||||
make docker-build IMG=egress-proxies-operator:dev
|
||||
docker run --rm egress-proxies-operator:dev --version
|
||||
# ae434a7167ec-dirty (matches git rev-parse --short=12 HEAD + untracked files)
|
||||
docker inspect egress-proxies-operator:dev --format '{{index .Config.Labels "org.opencontainers.image.revision"}}'
|
||||
# ae434a7167ec-dirty
|
||||
```
|
||||
@@ -0,0 +1,63 @@
|
||||
# Execution: GCP HTTP wire logging at V(5)
|
||||
|
||||
Plan: `docs/plans/2026-08-11-1838-gcp-http-wire-logging-v5.md`
|
||||
|
||||
- [x] Step 1 — `wireLogger` + `option.WithLogger` wiring in `internal/provider/gcp/gcp.go`
|
||||
- [x] Step 2 — Tests (`TestWireLogger_gatesAtV5`, `TestWireLogger_infoLandsAtV1`)
|
||||
- [x] Step 3 — Live verification at `--zap-log-level=5` (user, on cluster)
|
||||
- [x] Step 3b — Post-verification fix: drop auth records, elide huge fields
|
||||
- [ ] Step 4 — CHANGELOG entry (after live confirmation of 3b; batch with the
|
||||
two earlier pending entries: GCP V-logging, version stamp)
|
||||
|
||||
## Steps 1–2
|
||||
|
||||
Went exactly as planned — the whole feature is ~10 lines of production code
|
||||
because both halves already existed: the compute SDK logs full HTTP
|
||||
request/response records at slog Debug to an injectable logger, and
|
||||
`logr.ToSlogHandler` does the slog→logr bridging. The only real design
|
||||
content is the level shift (`base.V(1)` + slog-Debug's +4 = V(5)) and the
|
||||
startup warning line when V(5) is active (raw payloads include cloud-init
|
||||
user-data, which the curated V(2) logging deliberately hides).
|
||||
|
||||
Worth noting for future readers:
|
||||
|
||||
- `option.WithLogger` **disables** `GOOGLE_SDK_GO_LOGGING_LEVEL` for this
|
||||
client (documented SDK precedence) — `--zap-log-level` is now the only knob
|
||||
for GCP wire logs.
|
||||
- The V(5) check in `New` runs once at startup; that is sound because the zap
|
||||
level is fixed by flags at process start.
|
||||
- Added `TestWireLogger_infoLandsAtV1` beyond the plan's table — it pins the
|
||||
shift arithmetic from the other side (slog Info → V(1)), so a future logr
|
||||
mapping change would fail loudly.
|
||||
|
||||
Verified with:
|
||||
|
||||
```bash
|
||||
go test -race ./internal/provider/gcp/
|
||||
go build ./... && go test ./...
|
||||
```
|
||||
|
||||
## Step 3b — what live verification exposed, and the fix
|
||||
|
||||
Live V(5) output revealed two problems the plan missed:
|
||||
|
||||
1. **Security: the injected logger propagates into `cloud.google.com/go/auth`**,
|
||||
which logs its own token exchange (`auth.go:571/576`) — signed JWT
|
||||
assertion in the request, full bearer access token in the response. The
|
||||
plan's "auth token is safe" analysis only covered the compute client's
|
||||
request headers, not the auth library's own records. Fix: `wireLogger`
|
||||
now wraps the handler in a filter that drops every Debug record except
|
||||
the compute client's `"api request"`/`"api response"` (allowlist, so
|
||||
future SDK additions fail closed); Warn/Error still pass through.
|
||||
2. **Readability: GCP responses embed multi-KB blobs** (Shielded-VM UEFI
|
||||
dbx databases) that swamp the line. Fix: string fields >1KiB are elided
|
||||
to `[elided N bytes]` by default, recursively through payload
|
||||
maps/arrays. Opt-out via new manager flag
|
||||
`--gcp-wire-log-full-payloads` (threaded through a constructor closure
|
||||
in `cmd/main.go` → `gcp.NewWithWireOptions`; the `registry.Constructor`
|
||||
signature stays unchanged). Chosen by the user: elision on by default,
|
||||
verbatim available on demand. Auth records are dropped in both modes.
|
||||
|
||||
The filter/elision logic lives in `internal/provider/gcp/wirelog.go` with
|
||||
tests covering: auth-record drop (both modes), elision marker + small-field
|
||||
preservation, verbatim mode, and the original V(5) gating.
|
||||
@@ -0,0 +1,53 @@
|
||||
# Execution: Distilled Gitea Actions image-build workflow
|
||||
|
||||
Plan: `docs/plans/2026-08-11-1935-gitea-build-workflow.md`
|
||||
|
||||
- [x] Step 1 — Create `.gitea/workflows/build.yaml`
|
||||
- [x] Step 2 — Replace CLAUDE.md CI TODO with a CI/CD subsection
|
||||
- [x] Step 3 — Push branch + open MR
|
||||
- [ ] Step 4 — CHANGELOG entry (after the first successful run is confirmed)
|
||||
|
||||
## Steps 1–2 — workflow + CLAUDE.md
|
||||
|
||||
The workflow distills the house pattern from 9 sibling projects (survey in the plan)
|
||||
plus improvements none of them combine: an immutable `sha-<12>` tag, a lightweight
|
||||
test gate, `:latest` moving only on real tag pushes, and a `concurrency` group.
|
||||
|
||||
Work happened in a git worktree off `origin/main` so the main checkout (which had
|
||||
unrelated uncommitted changes) stayed untouched:
|
||||
|
||||
```bash
|
||||
git worktree add -b feat/gitea-build-workflow \
|
||||
"$SCRATCH/wt-build" origin/main
|
||||
```
|
||||
|
||||
Deviation from the plan's assumptions: while planning, `feat/proxy-operator` was
|
||||
still unmerged and the plan noted `main` lacked `internal/`/`test/`. By execution
|
||||
time `origin/main` had moved (`076bc66..f7000f7` — the proxy-operator MR merged),
|
||||
so the branch and its CI gate cover the full operator code.
|
||||
|
||||
Verified the workflow parses and the check-gate commands pass on this exact tree
|
||||
(ruby stands in for a YAML linter because the system python3 has no `yaml` module):
|
||||
|
||||
```bash
|
||||
ruby -ryaml -e "YAML.load_file('.gitea/workflows/build.yaml'); puts 'YAML OK'"
|
||||
go vet ./... && go build ./... && go test -short ./... # all packages ok
|
||||
```
|
||||
|
||||
## Step 3 — push + MR
|
||||
|
||||
Branch pushed and MR opened with `tea` (the worktree was then removed and the main
|
||||
checkout switched onto the branch so the files are visible locally):
|
||||
|
||||
```bash
|
||||
tea pr create --title "Add Gitea Actions image-build workflow" \
|
||||
--description "..." --base main --head feat/gitea-build-workflow
|
||||
# → https://gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/pulls/2
|
||||
```
|
||||
|
||||
Worth noting: the workflow itself cannot run end-to-end until (a) the MR merges
|
||||
(it only triggers on tags / manual dispatch, not branch pushes) and (b) the
|
||||
`REGISTRY_TOKEN` secret is created in this repo's Gitea settings (PAT with
|
||||
`write:package`). First real verification = manual dispatch with tag
|
||||
`manual-test`, expecting `manual-test` + `sha-…` in Packages and `:latest`
|
||||
untouched.
|
||||
38
docs/plans-executions/2026-08-11-2152-discovery-api-docs.md
Normal file
38
docs/plans-executions/2026-08-11-2152-discovery-api-docs.md
Normal file
@@ -0,0 +1,38 @@
|
||||
# Execution: Discovery API documentation
|
||||
|
||||
Plan: [2026-08-11-2152-discovery-api-docs.md](../plans/2026-08-11-2152-discovery-api-docs.md)
|
||||
|
||||
- [x] Step 1 — Write `docs/api.md` full API reference
|
||||
- [x] Step 2 — Add pointers in README and architecture.md
|
||||
|
||||
## Step 1 — docs/api.md
|
||||
|
||||
Wrote the full reference: base URL (in-cluster FQDN + port-forward), bearer
|
||||
auth, error envelope, configuration table, all five routes with schemas,
|
||||
status codes and curl examples, the selection/cooldown semantics section,
|
||||
an end-to-end curl walkthrough, and the in-memory/single-replica caveats.
|
||||
All facts were taken from the code, not from memory of prior docs.
|
||||
|
||||
Worth noting: the doc explicitly calls out two things no earlier doc
|
||||
stated for clients — that a report **without** a target on a targetless
|
||||
lease creates a *global* cooldown (blocking the proxy for everyone), and
|
||||
that `rate_limited` and `banned` currently behave identically. Both came
|
||||
straight from `internal/lease/store.go` and are easy to trip over.
|
||||
|
||||
## Step 2 — Pointers + verification
|
||||
|
||||
Added one-line links to the new doc in README's quickstart (above the curl
|
||||
block) and in `docs/architecture.md` §7. Verified the documented behavior
|
||||
against the tree rather than trusting the write-up:
|
||||
|
||||
```sh
|
||||
go build ./... && go test -short ./internal/discovery/ ./internal/lease/
|
||||
```
|
||||
|
||||
Both pass; a grep of `server_test.go` confirmed every documented status
|
||||
code and error code (`invalid_ttl`, `invalid_query`, `invalid_result`,
|
||||
`no_match`, `unknown_lease`, 201/204/401/404/409) is asserted by tests.
|
||||
|
||||
Worth noting: CHANGELOG entry deliberately deferred until the user
|
||||
confirms the docs read well, per the CHANGELOG convention's
|
||||
"once the user confirms it works" clause.
|
||||
55
docs/plans-executions/2026-08-11-2220-demo-scripts.md
Normal file
55
docs/plans-executions/2026-08-11-2220-demo-scripts.md
Normal file
@@ -0,0 +1,55 @@
|
||||
# Execution: Demo scripts
|
||||
|
||||
Plan: [2026-08-11-2220-demo-scripts.md](../plans/2026-08-11-2220-demo-scripts.md)
|
||||
|
||||
- [x] Step 0 — Branch `feat/demo-scripts` + plan commit
|
||||
- [x] Step 1 — `docs/demo/show-egress-ips.sh`
|
||||
- [x] Extra (added iteratively, not in the original plan) — proxy-creation scripts
|
||||
|
||||
## Step 0 + Step 1
|
||||
|
||||
Branched off `main`, committed the plan alone, then wrote
|
||||
`docs/demo/show-egress-ips.sh`: takes `BASE_URL` as its argument, lists
|
||||
`/v1/proxies` (bearer auth via optional `TOKEN` env), probes each healthy
|
||||
proxy with `curl -x http://ip:port` against an IP-echo site
|
||||
(`IP_ECHO_URL`, default ipify JSON), banners which proxy each request goes
|
||||
through, skips unhealthy ones, and ends with a probed/skipped/failed
|
||||
summary. Per user request the script was left uncommitted for iteration
|
||||
and no verification beyond `bash -n` was run.
|
||||
|
||||
## Extra — create-kubernetes-proxies.sh, create-gcp-proxies.sh
|
||||
|
||||
Added on the same branch before the first commit:
|
||||
|
||||
- `create-kubernetes-proxies.sh <count>` — creates
|
||||
`proxy-kubernetes-demo-1..N` with the kubernetes provider, spec taken
|
||||
from `config/samples/proxy_kubernetes.yaml`, applied via
|
||||
`kubectl apply -f -` heredocs. Optional `NAMESPACE` env.
|
||||
- `create-gcp-proxies.sh <count>` — creates `proxy-gcp-demo-1..N` from the
|
||||
user-supplied gcp-eu manifest (e2-micro, Ubuntu 24.04, Squid
|
||||
cloud-init), each with a zone picked randomly from a hardcoded list of
|
||||
27 EU zones so the fleet gets egress IPs from different locations.
|
||||
Env overrides: `ZONES`, `GCP_PROVIDER` (default `gcp-eu`), `NAMESPACE`.
|
||||
|
||||
## Extra — run-demo.sh, show-egress-ips-table.sh
|
||||
|
||||
Second round of iterative additions:
|
||||
|
||||
- `run-demo.sh` — tmux demo driver: 2x2 tiled grid with the egress-IP
|
||||
table looping every 10s inside a netshoot pod (script `kubectl cp`'d
|
||||
into the pod, `BASE_URL` set to the in-cluster Service FQDN),
|
||||
`watch -n3 kubectl get px`, and both create scripts auto-running with
|
||||
`COUNT` proxies each (default 4). Env knobs: `COUNT`, `SESSION`,
|
||||
`NETSHOOT_POD`, `DEMO_DIR` (defaults to the script's own dir).
|
||||
- `show-egress-ips-table.sh` — condensed one-line-per-proxy variant of
|
||||
`show-egress-ips.sh` sized for a tmux pane: PROXY / ENDPOINT /
|
||||
LOCATION (zone→geo attribute fallback) / EGRESS-IP columns, with
|
||||
`(unhealthy)` and `FAILED` inline instead of verbose output. The
|
||||
verbose script stays for standalone use; the demo driver uses the
|
||||
table variant.
|
||||
|
||||
Worth noting: beyond the user's sample manifest, the gcp script also
|
||||
writes the picked zone into `attributes.zone`, so the discovery API
|
||||
exposes each proxy's location and leases can select on it. The zone list
|
||||
is static — if a project lacks quota in some region, `ZONES` narrows the
|
||||
pool; nothing validates zones against the live project.
|
||||
@@ -22,7 +22,7 @@ spec change deletes and recreates the VM. No in-place update logic.
|
||||
| API group | `crawl.example.com`, `v1alpha1`, kind `Proxy`, **namespaced** |
|
||||
| Git flow | Scaffold commit, then work on `feat/proxy-operator`, MR via `tea`, no merge/delete from CLI |
|
||||
| kubebuilder | `go install sigs.k8s.io/kubebuilder/v4/cmd/kubebuilder@v4.15.0` |
|
||||
| Mock health | Mock provider runs a **real in-process HTTP CONNECT proxy** per instance, so healthchecks genuinely pass and the kind demo is truly end-to-end |
|
||||
| Local/CI provider | **Kubernetes pod provider**, not an in-memory mock: creates real `ubuntu/squid` pods in-cluster. Revised mid-build — see Step 3 |
|
||||
| Verification | vet + unit + envtest, then a throwaway kind cluster running the README quickstart, then delete it |
|
||||
|
||||
### Verified environment — no version substitutions needed
|
||||
@@ -38,7 +38,7 @@ omitting the substitutions section.
|
||||
|
||||
### Milestone order — each ends green on `go build ./... && go vet ./...`
|
||||
|
||||
1. Scaffold + pins → 2. `api/v1alpha1` + CEL → 3. `internal/provider` + mock →
|
||||
1. Scaffold + pins → 2. `api/v1alpha1` + CEL → 3. `internal/provider` + kubernetes-pod →
|
||||
4. reconciler + envtest → 5. health engine → 6. lease + discovery → 7. GCP provider →
|
||||
8. orphan GC + metrics → 9. `cmd/main.go` wiring + `config/` → 10. docs + kind run.
|
||||
|
||||
@@ -64,8 +64,9 @@ plan to `docs/plans/2026-08-07-1747-proxy-operator.md` per CLAUDE.md. **Commit t
|
||||
untouched scaffold on its own** so every later diff is reviewable.
|
||||
|
||||
Post-scaffold hand-edits: `CONTROLLER_TOOLS_VERSION ?= v0.21.0` in the Makefile (CEL
|
||||
emission at the 1.36 API level); add a `run-mock` target; delete the scaffolded
|
||||
`.github/workflows/` (the remote is Gitea).
|
||||
emission at the 1.36 API level); add a `run-dev` target wired to a sample
|
||||
`--providers-config` (Step 10); delete the scaffolded `.github/workflows/` (the
|
||||
remote is Gitea).
|
||||
|
||||
---
|
||||
|
||||
@@ -120,12 +121,15 @@ health, discovery, and the hash.
|
||||
```
|
||||
internal/provider/{provider,errors,name,config,metrics}.go
|
||||
internal/provider/registry/registry.go # type→constructor — SEPARATE package
|
||||
internal/provider/{mock,gcp}/
|
||||
internal/provider/{kubernetes,gcp}/
|
||||
```
|
||||
|
||||
**Import-cycle trap:** a registry inside `internal/provider` would have to import
|
||||
`internal/provider/mock`, which imports `internal/provider`. `init()` self-registration
|
||||
is banned by CLAUDE.md, so the registry goes in its own leaf-importing package.
|
||||
`internal/provider/kubernetes` (or `.../gcp`), both of which import `internal/provider`
|
||||
for the interface. `init()` self-registration is banned by CLAUDE.md, so the registry
|
||||
goes in its own leaf-importing package, and its `Build` function takes the
|
||||
type→constructor map as a parameter instead — see Step 2's execution log entry for why
|
||||
this ended up better than a package-level map even beyond avoiding the cycle.
|
||||
|
||||
`Instance` needs **two fields the spec omits**, or orphan GC is unimplementable:
|
||||
`UID string` (from the label, for the liveness match) and `CreatedAt time.Time` (for
|
||||
@@ -157,23 +161,67 @@ same 80 bits. 80 bits → birthday collision at ~2^40 objects against a fleet of
|
||||
|
||||
---
|
||||
|
||||
## Step 3 — Mock provider (`internal/provider/mock/`)
|
||||
## Step 3 — Kubernetes pod provider (`internal/provider/kubernetes/`)
|
||||
|
||||
**State is a pure function of an injectable clock — no background timers.** `Get`/
|
||||
`ListByTag` derive state from `createdAt`/`deletedAt` vs `now()`:
|
||||
`< provisionDelay` → Provisioning; else Running; `deletedAt` set and `< deleteDelay` →
|
||||
Terminated; beyond that → purged, `ErrNotFound`. Deterministic under a fake clock,
|
||||
correct under the real one, and no goroutine lifecycle to leak.
|
||||
**Revised after Step 3 was first built as an in-memory mock provider** (state-machine
|
||||
simulation + a hand-rolled CONNECT proxy on a shared, refcounted local listener). The
|
||||
user found that too far from the real system to build confidence in, and didn't need
|
||||
tests to be fast enough to justify the complexity it cost — a real `kind`-cluster
|
||||
verification pass "once in a while" is an acceptable trade for tests that actually
|
||||
look like the final product. Full narrative of the reversal is in
|
||||
`docs/plans-executions/2026-08-07-1747-proxy-operator.md`; this section describes the
|
||||
replacement, which is what actually gets built for local dev/CI going forward. No
|
||||
in-memory provider remains in the tree — GCP is now the only other provider, per the
|
||||
spec's original two-provider scope.
|
||||
|
||||
**Real proxy listener (the user's decision):** when a record first reports `Running`,
|
||||
lazily start an `http.Server` on `127.0.0.1:0` implementing HTTP `CONNECT` tunnelling
|
||||
(hijack + bidirectional `io.Copy`) plus plain-HTTP forwarding, and report `127.0.0.1`
|
||||
plus the real listener port. `Delete` shuts it down. This is what makes the health
|
||||
engine's probe a genuine CONNECT through a real proxy, so the kind quickstart actually
|
||||
reaches Ready and leasable. One `http.Server` per instance, bounded by fleet size.
|
||||
**What it does:** `Create` creates a `corev1.Pod` running a proxy container in the
|
||||
same cluster (and same namespace as the owning Proxy CR — `req.Namespace`); `Get`
|
||||
reads the Pod's phase/IP; `Delete` deletes it (tolerating NotFound); `ListByTag` lists
|
||||
Pods by the standard `LabelManaged`/`LabelUID` labels, unscoped by namespace (the
|
||||
operator's RBAC needs cluster-scoped Pod permissions — see RBAC note below).
|
||||
|
||||
Fault injection: config-driven `failNextCreates`/`failWith` for the demo, plus
|
||||
`InjectCreateFailures(n int, class error)` for tests. `Create` is idempotent by name.
|
||||
**Proxy software: `ubuntu/squid`** (Canonical's actively maintained LTS image on
|
||||
Docker Hub, verified before picking it — 50M+ pulls, updated the same day this
|
||||
decision was made), not a hand-rolled proxy. It's a public image, so `kind` nodes
|
||||
pull it directly; no build/load step needed for the quickstart. Squid's config
|
||||
(`http_port <req.Port>`, permissive ACL) is generated in Go and injected via an env
|
||||
var the container's command writes to `/etc/squid/squid.conf` before exec'ing squid —
|
||||
no separate ConfigMap object, so there's still only one Kubernetes object per proxy
|
||||
instance to create, track, and clean up.
|
||||
|
||||
**providerID format: `<namespace>/<podName>`** (parseable with
|
||||
`k8s.io/client-go/tools/cache.SplitMetaNamespaceKey`), so `Get`/`Delete` are
|
||||
self-contained without needing to re-derive the namespace — the same reasoning as the
|
||||
GCP provider's zone-qualified providerID in Step 8.
|
||||
|
||||
**Pod naming:** reuses `provider.NameFromUID` unchanged — the same deterministic name
|
||||
satisfies Kubernetes Pod naming rules (`^[a-z0-9]([-a-z0-9]*[a-z0-9])?$`, ≤253 chars)
|
||||
with room to spare.
|
||||
|
||||
**State mapping:** Pod phase `Pending`, or `Running` with no PodIP yet → Provisioning
|
||||
(never publish an empty IP); `Running` with a PodIP → Running; `Succeeded`/`Failed`/
|
||||
`Unknown` → Terminated (the reconciler treats Stopped and Terminated identically —
|
||||
delete and recreate, cattle not pets — so collapsing three failure-ish phases into one
|
||||
is enough).
|
||||
|
||||
**Client:** built internally via `ctrl.GetConfig()` (auto-detects in-cluster config,
|
||||
falls back to the local kubeconfig otherwise), not threaded through the registry
|
||||
`Constructor` signature — this is what makes `make run` against a local `kind`
|
||||
cluster and running in-cluster use the exact same code path with no provider-specific
|
||||
wiring in `cmd/main.go`.
|
||||
|
||||
**Testing, given envtest can't schedule real Pods** (no kubelet — a Pod created
|
||||
against `envtest`'s API server just sits `Pending` forever): `internal/provider/kubernetes`
|
||||
itself is unit-tested against `sigs.k8s.io/controller-runtime/pkg/client/fake` — real
|
||||
Pod objects, real client interface, fully exercises Create/Get/Delete/ListByTag
|
||||
logic and the Pod-construction function in isolation, just without a kubelet actually
|
||||
starting a container. Real end-to-end proof (does a probe actually tunnel through a
|
||||
real Squid pod) only happens against a real `kind` cluster, in the Verification
|
||||
section — which is exactly what the user asked for. Step 4's reconciler tests use a
|
||||
small `Provider`-interface stub defined directly in the controller test file (a
|
||||
handful of lines, not a package) for exercising the state-machine's branching logic —
|
||||
categorically simpler than what mock.Provider was, since it has no config format, no
|
||||
fault-injection surface, and exists only inside test code.
|
||||
|
||||
---
|
||||
|
||||
@@ -343,7 +391,10 @@ pods would refuse connections while still being Service endpoints. The 1-replica
|
||||
constraint comes from lease state being per-process, which the spec already accepts —
|
||||
both facts go in the README caveats.
|
||||
|
||||
stdlib `http.ServeMux` with Go 1.22 method+wildcard patterns:
|
||||
stdlib `http.ServeMux` using its method+wildcard patterns (`"GET /path"`,
|
||||
`"/{id}"` + `r.PathValue`) — a stdlib feature available since Go 1.22, used
|
||||
here so no third-party router is needed; the project itself stays on the
|
||||
pinned Go 1.26:
|
||||
`GET /v1/proxies`, `POST /v1/leases`, `DELETE /v1/leases/{id}`,
|
||||
`POST /v1/leases/{id}/report`, plus unauthenticated `GET /healthz`.
|
||||
Middleware outermost-first: recover → request-log → `MaxBytesReader(64KiB)` → bearer
|
||||
@@ -425,14 +476,17 @@ registry → manager → `mgr.Add` health engine, GC, lease expiry loop, discove
|
||||
`SetupWithManager`. Contexts from `ctrl.SetupSignalHandler()` throughout.
|
||||
|
||||
RBAC markers: proxies CRUD + status + finalizers, secrets get/list/watch, events
|
||||
create/patch. `config/`: providers ConfigMap mount, `DISCOVERY_TOKEN` from a Secret,
|
||||
containerPort 8090 + Service. Samples: `proxy_mock.yaml`, `proxy_gcp.yaml`,
|
||||
`proxy_external.yaml`, `providers-config.yaml`.
|
||||
create/patch, and (for the kubernetes-pod provider) pods get/list/watch/create/delete
|
||||
— cluster-scoped, since `ListByTag` enumerates across namespaces. `config/`: providers
|
||||
ConfigMap mount, `DISCOVERY_TOKEN` from a Secret, containerPort 8090 + Service.
|
||||
Samples: `proxy_kubernetes.yaml`, `proxy_gcp.yaml`, `proxy_external.yaml`,
|
||||
`providers-config.yaml`.
|
||||
|
||||
`docs/architecture.md`: components table, ASCII data-flow diagram, and a **Decisions**
|
||||
section covering the channel-vs-patch choice, replacement-polls-to-NotFound, base32
|
||||
naming, discovery without leader election, single-mutex lease store, mock-runs-a-real-
|
||||
proxy, leader-handover health seeding, the latency-suppression refinement, the
|
||||
naming, discovery without leader election, single-mutex lease store, the mock→
|
||||
kubernetes-pod-provider revision (why, and why `ubuntu/squid` over a hand-rolled
|
||||
proxy), leader-handover health seeding, the latency-suppression refinement, the
|
||||
`lastHealthCheckTime` semantics, and the logr-not-slog deviation (CLAUDE.md says
|
||||
`slog`, but `log.FromContext(ctx)` returns logr inside controller paths — noted, not
|
||||
silently ignored).
|
||||
@@ -448,16 +502,17 @@ note confirming no substitutions were needed. Then a `CHANGELOG.md` entry with a
|
||||
|
||||
## Step 11 — Tests
|
||||
|
||||
**envtest** (`internal/controller/`), mock provider + a fake `HealthSnapshotter`,
|
||||
intervals shrunk to 50–200 ms, whole suite behind `testing.Short()`:
|
||||
Managed→Ready; spec change → old mock instance gone and providerID changed; delete →
|
||||
finalizer runs and instance removed; External → Ready on first health pass, no
|
||||
finalizer; injected quota → `Provisioned=False/QuotaExceeded` and phase *not* Failed;
|
||||
permanent error → Failed and no further provider calls; adopt (strip annotation →
|
||||
restored, providerID unchanged); and the **CEL cases only a real API server can test** —
|
||||
mode/provider mutation rejected, Managed-without-provider, External-without-endpoint,
|
||||
cloudInit both/neither, and `healthCheck` omitted → nested defaults materialized (the
|
||||
`default={}` assertion).
|
||||
**envtest** (`internal/controller/`), a small in-test stub `Provider` (not the real
|
||||
kubernetes-pod provider — envtest has no kubelet, so a real Pod never leaves Pending)
|
||||
plus a fake `HealthSnapshotter`, intervals shrunk to 50–200 ms, whole suite behind
|
||||
`testing.Short()`: Managed→Ready; spec change → old stub instance gone and
|
||||
providerID changed; delete → finalizer runs and instance removed; External → Ready
|
||||
on first health pass, no finalizer; injected quota → `Provisioned=False/QuotaExceeded`
|
||||
and phase *not* Failed; permanent error → Failed and no further provider calls; adopt
|
||||
(strip annotation → restored, providerID unchanged); and the **CEL cases only a real
|
||||
API server can test** — mode/provider mutation rejected, Managed-without-provider,
|
||||
External-without-endpoint, cloudInit both/neither, and `healthCheck` omitted → nested
|
||||
defaults materialized (the `default={}` assertion).
|
||||
|
||||
**Action-table unit tests** — the highest-value tests in the repo: `fake` client with
|
||||
`WithStatusSubresource`, calling `Reconcile` directly, table-driven over every row of
|
||||
@@ -466,12 +521,15 @@ client runs neither CEL nor defaulting — that's what the envtest CEL cases cov
|
||||
|
||||
**Units:** name derivation (idempotency, `^proxy-[a-z2-7]{16}$`, 10k-UID distinctness);
|
||||
`Class()` mapping + `errors.Is`/`errors.As` through the multi-unwrap; config loading;
|
||||
mock state at the delay boundary with a fake clock; `buildInsertRequest` field-by-field
|
||||
+ GCP error classification + RUNNING-without-IP; `computePhase` truth table; `SpecHash`
|
||||
stability *and* sensitivity; lease store (capacity, `MaxLeases=0`, least-loaded with
|
||||
latency tie-break, cooldown with/without target, report on an expired-but-retained
|
||||
lease, concurrent acquire under `-race` never exceeding `MaxLeases`); discovery
|
||||
handlers over `httptest` + fake reader + real store; health thresholds against a real
|
||||
kubernetes-pod provider Create/Get/Delete/ListByTag against
|
||||
`sigs.k8s.io/controller-runtime/pkg/client/fake` (real Pod objects, real client
|
||||
interface, no kubelet needed for this level) plus pure tests of the generated Squid
|
||||
config and Pod spec; `buildInsertRequest` field-by-field + GCP error classification +
|
||||
RUNNING-without-IP; `computePhase` truth table; `SpecHash` stability *and*
|
||||
sensitivity; lease store (capacity, `MaxLeases=0`, least-loaded with latency
|
||||
tie-break, cooldown with/without target, report on an expired-but-retained lease,
|
||||
concurrent acquire under `-race` never exceeding `MaxLeases`); discovery handlers
|
||||
over `httptest` + fake reader + real store; health thresholds against a real
|
||||
CONNECT-capable `httptest` proxy stub.
|
||||
|
||||
Everything runs with `-race`.
|
||||
@@ -485,19 +543,20 @@ go vet ./... && make test && make build # unit + envtest, -race
|
||||
|
||||
kind create cluster --name proxy-operator-demo
|
||||
make install
|
||||
make run-mock & # --providers-config hack/providers-mock.yaml
|
||||
kubectl apply -f config/samples/proxy_mock.yaml
|
||||
kubectl get px -w # expect Ready with an IP
|
||||
make run-dev & # --providers-config hack/providers-dev.yaml
|
||||
kubectl apply -f config/samples/proxy_kubernetes.yaml
|
||||
kubectl get px -w # expect Ready with an IP (a real squid Pod)
|
||||
curl -s 'localhost:8090/v1/proxies?healthy=true' | jq
|
||||
curl -s -XPOST localhost:8090/v1/leases -d '{"selector":{"geo":"eu"},"ttlSeconds":300}' | jq
|
||||
curl -s -XPOST localhost:8090/v1/leases/<id>/report -d '{"result":"rate_limited","target":"example.com"}'
|
||||
curl -si -XDELETE localhost:8090/v1/leases/<id> # 204, and 204 again
|
||||
kubectl delete -f config/samples/proxy_mock.yaml # finalizer runs, object goes
|
||||
kubectl delete -f config/samples/proxy_kubernetes.yaml # finalizer runs, Pod is deleted
|
||||
kind delete cluster --name proxy-operator-demo
|
||||
```
|
||||
|
||||
Success bar: a competent SRE clones the repo, follows the README, and holds a lease on a
|
||||
healthy mock proxy in under 10 minutes.
|
||||
healthy proxy — a real Squid pod running in their own `kind` cluster — in under 10
|
||||
minutes.
|
||||
|
||||
Then commit on `feat/proxy-operator`, push with `-u`, open the MR with
|
||||
`tea pr create --base main --head feat/proxy-operator`, print the URL. No merging or
|
||||
|
||||
128
docs/plans/2026-08-08-1335-lean-scaffold-cleanup.md
Normal file
128
docs/plans/2026-08-08-1335-lean-scaffold-cleanup.md
Normal file
@@ -0,0 +1,128 @@
|
||||
# Plan: Lean-down cleanup — strip non-goal scaffold from the application footprint
|
||||
**Created:** 2026-08-08 13:35
|
||||
|
||||
## Context
|
||||
|
||||
The user asked two things: (1) explain why webhook-related pieces existed at all when
|
||||
the spec ([docs/prompts/__initial-prompt.md](docs/prompts/__initial-prompt.md) §12)
|
||||
says "do NOT build: admission webhooks, cert-manager wiring", and (2) audit all
|
||||
completed work (Steps 0–3) for anything extra, so the project starts as lean as
|
||||
possible before Step 4 (reconciler) begins.
|
||||
|
||||
**The answer to (1), already given in conversation and to be recorded in the
|
||||
execution log:** `kubebuilder init` unconditionally generates webhook machinery for
|
||||
every project. At Step 0 the scaffold was deliberately committed untouched as a
|
||||
reviewable baseline, with only `.github/` stripped; spec §3's "config/
|
||||
(scaffold-generated, kept working)" was read as license to keep the rest. That was a
|
||||
process gap — §12's non-goals deserved an active pruning pass immediately after the
|
||||
baseline commit, especially for parts that actually *did* something (main.go started
|
||||
a real webhook server; e2e installed cert-manager). The active parts were already
|
||||
removed in commit `7700358` after the user noticed; this plan removes what remains.
|
||||
|
||||
**Could `init` have been told to skip it? No — verified against the v4.15.0 binary
|
||||
(`kubebuilder init --help`), not from memory.** Its full flag surface is
|
||||
domain/repo/owner/license/multigroup/namespaced/fetch-deps/skip-go-version-check/
|
||||
project-version/plugins; nothing subtracts features. Plugins are purely additive
|
||||
(helm, grafana, deploy-image, autoupdate — no "minimal" plugin), and while webhook
|
||||
*code* only appears via `create webhook` (never run here), the baseline *plumbing*
|
||||
(main.go webhook server, commented kustomize blocks, cert-manager in e2e utils,
|
||||
prometheus/network-policy dirs) is emitted unconditionally so later `create webhook`
|
||||
runs have anchors. Scaffold-then-prune is the only supported path to a lean
|
||||
baseline. (`--namespaced` was the one arguably-applicable flag, but it conflicts
|
||||
with the kubernetes-pod provider's cluster-wide `ListByTag` for orphan GC —
|
||||
cluster-scoped was correct.) Record this in the execution log so the next project
|
||||
bootstrap knows to plan a pruning pass at scaffold time.
|
||||
|
||||
**Scope, per the user's explicit direction:** "tooling around the project is cool, i
|
||||
just want the application code produced to start as lean as possible." So developer
|
||||
tooling stays untouched — `.golangci.yml`, `.custom-gcl.yml`, `.devcontainer/`,
|
||||
`AGENTS.md`, Makefile `lint`/`docker-buildx`/`build-installer` targets, and
|
||||
`.claude/settings.json` are all explicitly KEPT. The cleanup targets only the
|
||||
application and its deployed footprint: `config/` manifests that `make deploy`
|
||||
would apply or that exist solely to serve never-to-be-built features.
|
||||
|
||||
**Code-level audit result (part of this task's deliverable, no action needed):**
|
||||
Steps 1–3's Go code contains nothing beyond spec that isn't a justified,
|
||||
already-logged deviation (`Instance.UID`/`CreatedAt` for orphan GC, `MaxLeases
|
||||
*int32`, `HealthCheck default={}`, registry-as-parameter, kubernetes provider
|
||||
replacing mock per user decision). Trivial extras (`shortName=px`,
|
||||
`MaxProperties=32` on attributes) are harmless and stay. `metrics_auth_role*.yaml` /
|
||||
`metrics_reader_role.yaml` are genuinely used by the secure-metrics filter and the
|
||||
e2e metrics test — they stay.
|
||||
|
||||
## Scope refinements from user review
|
||||
|
||||
- **KEEP `config/prometheus/`** entirely (user: monitoring manifests stay),
|
||||
including `monitor_tls_patch.yaml` and the commented `#- ../prometheus` enable
|
||||
line in the default kustomization.
|
||||
- **KEEP the paired metrics-TLS plumbing** for coherence with the kept prometheus
|
||||
TLS patch: `config/default/cert_metrics_manager_patch.yaml`, its commented
|
||||
`[METRICS-WITH-CERTS]` reference, and the *metrics-certs/ServiceMonitor halves*
|
||||
of the commented replacements block. Removing half of a pair would leave
|
||||
dangling comment references — mess, not lean.
|
||||
- **REMOVE `config/network-policy/`** (user: "we do not need any network policies
|
||||
at the moment").
|
||||
- **KEEP all RBAC manifests** (user: "i want to keep manifests relevant to
|
||||
rbacs") — including the `proxy_admin/editor/viewer` helper ClusterRoles
|
||||
originally slated for removal.
|
||||
|
||||
## Removals (all in `config/`)
|
||||
|
||||
1. **`config/network-policy/`** (2 files: kustomization.yaml,
|
||||
allow-metrics-traffic.yaml) + the commented `#- ../network-policy` line and its
|
||||
`[NETWORK POLICY]` banner in `config/default/kustomization.yaml`.
|
||||
|
||||
2. **`config/default/kustomization.yaml`** — strip the *webhook-only* parts: the
|
||||
commented `#- ../webhook` and `#- ../certmanager` resource lines with their
|
||||
banners; the commented `manager_webhook_patch.yaml` patch reference; and the
|
||||
webhook halves of the commented replacements block (`serving-cert` Certificate
|
||||
sources targeting Validating/Mutating WebhookConfiguration cainjection, the
|
||||
conversion-webhook block, and the
|
||||
`+kubebuilder:scaffold:crdkustomizecainjectionns`/`...name` markers — anchors
|
||||
only for `kubebuilder create webhook`, which will never run here). The
|
||||
metrics-certs/ServiceMonitor replacement halves stay (see scope refinements).
|
||||
|
||||
3. **`config/crd/kustomization.yaml`** — strip the two commented `[WEBHOOK]` blocks
|
||||
(conversion-webhook patches and the `configurations:` reference) and the
|
||||
`+kubebuilder:scaffold:crdkustomizewebhookpatch` marker. **Keep** the
|
||||
`+kubebuilder:scaffold:crdkustomizeresource` marker (one line; anchors
|
||||
`kubebuilder create api`, which could legitimately run again).
|
||||
|
||||
4. **`config/crd/kustomizeconfig.yaml`** — teaches kustomize how to rewrite webhook
|
||||
conversion service references; only consumer was the commented block in (3).
|
||||
|
||||
`config/rbac/` is untouched: the operator's own role/bindings, the metrics-auth
|
||||
roles, *and* the `proxy_admin/editor/viewer` helper ClusterRoles all stay per the
|
||||
user's direction.
|
||||
|
||||
## Execution steps
|
||||
|
||||
1. Save this plan into the repo per CLAUDE.md: `docs/plans/$(date
|
||||
"+%Y-%m-%d-%H%M")-lean-scaffold-cleanup.md`, committed on its own before the
|
||||
cleanup work starts.
|
||||
2. Delete the files listed above (`git rm`); edit the two kustomization.yaml files.
|
||||
3. Verify:
|
||||
- `bin/kustomize build config/default` renders cleanly (proves no dangling
|
||||
references; kustomize is already in `bin/` from the scaffold).
|
||||
- `bin/kustomize build config/crd` renders cleanly.
|
||||
- `go build ./... && go vet ./...` and the `-tags=e2e` variants stay clean.
|
||||
- `make test` stays green.
|
||||
4. Append an execution-log section to
|
||||
[docs/plans-executions/2026-08-07-1747-proxy-operator.md](docs/plans-executions/2026-08-07-1747-proxy-operator.md)
|
||||
covering: the "why webhooks existed" explanation (scaffold origin + the Step 0
|
||||
process gap), the audit's clean bill for the Go code, the keep-tooling scope
|
||||
decision, and the exact removals.
|
||||
5. Single commit on `feat/proxy-operator` with the plan-file commit preceding it.
|
||||
|
||||
## Verification
|
||||
|
||||
```bash
|
||||
bin/kustomize build config/default > /dev/null && echo default-ok
|
||||
bin/kustomize build config/crd > /dev/null && echo crd-ok
|
||||
go build ./... && go vet ./...
|
||||
go build -tags=e2e ./... && go vet -tags=e2e ./...
|
||||
make test
|
||||
```
|
||||
|
||||
All must pass with output identical in substance to pre-cleanup (same tests, same
|
||||
coverage numbers).
|
||||
87
docs/plans/2026-08-11-1742-gcp-provider-verbose-logging.md
Normal file
87
docs/plans/2026-08-11-1742-gcp-provider-verbose-logging.md
Normal file
@@ -0,0 +1,87 @@
|
||||
# Plan: Verbose (V-level) logging in the GCP provider
|
||||
|
||||
**Created:** 2026-08-11 17:42
|
||||
|
||||
## Context
|
||||
|
||||
Debugging GCP provisioning is currently blind: the operator has zero `.V(n)` calls
|
||||
anywhere, so `--zap-log-level=debug` (or any numeric level) reveals nothing about
|
||||
what the GCP provider is doing — which API calls it makes, with what parameters,
|
||||
and what came back. The goal: with debug/V-level logging enabled, see the details
|
||||
of every GCP Compute API call (Insert/Get/Delete/AggregatedList) including a
|
||||
summary of the response; with default `info` level, the provider stays as quiet
|
||||
as today.
|
||||
|
||||
## Approach (the "how")
|
||||
|
||||
**Logger source — context-carried, not injected.** Provider methods all take
|
||||
`ctx`, and the reconciler already builds a per-request logger
|
||||
(`logf.FromContext(ctx)` in `internal/controller/proxy_controller.go:117`) that
|
||||
carries the proxy's name/namespace. The GCP provider will do
|
||||
`log := logf.FromContext(ctx).WithName("gcp").WithValues("provider", p.name)` at
|
||||
the top of each public method. Zero wiring changes (no registry/constructor/struct
|
||||
changes), and every provider log line automatically inherits the reconcile
|
||||
context (which Proxy triggered it). Calls from the GC sweeper inherit its
|
||||
`orphan-gc` logger name the same way.
|
||||
|
||||
**Verbosity scheme** (logr convention: `.Info()` = V(0), `debug` flag = V(1)):
|
||||
|
||||
- **V(1)** — one line per GCP API call, after it returns: operation, identifying
|
||||
params, outcome. Examples:
|
||||
- `Create`: `"GCP insert instance"` with `zone`, `name`, `machineType`,
|
||||
`image`, `opName` (currently discarded at gcp.go:117 — capture it, it's the
|
||||
only handle for correlating with GCP's operation log), plus a line for the
|
||||
409-already-exists path.
|
||||
- `Get`: `"GCP get instance"` with `zone`, `name`, `status`, mapped `state`, `ip`.
|
||||
- `Delete`: `"GCP delete instance"` with `zone`, `name`, `opName`, and the
|
||||
404-treated-as-success path.
|
||||
- `ListByTag`: `"GCP aggregated list"` with `filter`, `count`.
|
||||
- Error paths at V(1) too: log the raw classification (HTTP status / reason
|
||||
from `googleapi.Error`) before it's wrapped, since the wrapped error the
|
||||
reconciler sees is coarser.
|
||||
- **V(2)** — request/response detail: full curated insert-request summary
|
||||
(network, networkTag, diskSizeGB, port, labels, `cloudInitBytes` = `len`),
|
||||
per-instance lines in `ListByTag` (id, state, uid, age).
|
||||
|
||||
**Curated fields, never raw proto dumps.** `CreateRequest.CloudInit` is resolved
|
||||
user-data possibly sourced from a Secret, and it lands in the insert request's
|
||||
metadata — so logging the request proto wholesale would leak it. Log named safe
|
||||
fields only; for cloud-init, log only its byte length. This is a hard rule, and
|
||||
a test asserts it.
|
||||
|
||||
**Where the calls live: the `Provider` methods in
|
||||
`internal/provider/gcp/gcp.go`** (Create/Get/Delete/ListByTag), not in
|
||||
`realInstances` (deliberately untested by design, gcp.go:86) and not an HTTP
|
||||
round-tripper (would log auth headers/user-data, unredactable). The
|
||||
`instancesAPI` fake seam (`newWithAPI`, gcp.go:96) keeps everything testable.
|
||||
To surface `opName`, change `Provider.Create`/`Delete` to capture the string
|
||||
their `instancesAPI` calls already return instead of discarding it.
|
||||
|
||||
## Files to change
|
||||
|
||||
- `internal/provider/gcp/gcp.go` — add `logf` import; V(1)/V(2) logging in
|
||||
`Create`, `Get`, `Delete`, `ListByTag`; capture opNames. Only file with
|
||||
production changes.
|
||||
- `internal/provider/gcp/gcp_test.go` — new table-driven test
|
||||
`TestLogging_verbosity` (name TBD per house `Test<Function>_<scenario>`
|
||||
style): inject a capturing logger via `logf.IntoContext(ctx, funcr.New(...))`
|
||||
(`github.com/go-logr/logr/funcr`, logr already a direct dep), assert:
|
||||
- at V(1): expected message + keys per operation (incl. opName),
|
||||
- at V(0): nothing logged,
|
||||
- **cloud-init content never appears in any log output** (grep the captured
|
||||
lines for a sentinel string placed in `CloudInit`).
|
||||
- No changes to `provider.Provider` interface, registry, `cmd/main.go`,
|
||||
manifests, or the kubernetes provider (it can copy this pattern later).
|
||||
|
||||
## Verification
|
||||
|
||||
```bash
|
||||
go test -race ./internal/provider/gcp/...
|
||||
go build ./...
|
||||
```
|
||||
|
||||
Optional live check: run the manager with `--zap-log-level=2` against the GCP
|
||||
project and confirm insert/get lines appear during a Proxy reconcile, and that
|
||||
`--zap-log-level=info` stays quiet.
|
||||
|
||||
Also append a CHANGELOG.md entry per house convention once confirmed working.
|
||||
75
docs/plans/2026-08-11-1802-bake-commit-version.md
Normal file
75
docs/plans/2026-08-11-1802-bake-commit-version.md
Normal file
@@ -0,0 +1,75 @@
|
||||
# Plan: Bake the git commit into the operator binary and log it at startup
|
||||
|
||||
**Created:** 2026-08-11 18:02
|
||||
|
||||
## Context
|
||||
|
||||
There is no versioning yet, and the image tag (`egress-proxies-operator:dev`,
|
||||
`imagePullPolicy: IfNotPresent`) says nothing about what code is actually
|
||||
running. The user wants the commit hash baked into the image at build time,
|
||||
and — the key requirement — the binary itself must know it and print it into
|
||||
the log stream during initialization, so `kubectl logs | head` answers "which
|
||||
version is running". Docker builds cannot use Go's automatic VCS stamp because
|
||||
`.dockerignore` excludes `.git` (correctly — re-including it would bust layer
|
||||
caching), so the hash must travel git → Makefile → `--build-arg` → `-ldflags -X`.
|
||||
|
||||
## Implementation
|
||||
|
||||
**1. New package `internal/version`** (`version.go` + `version_test.go`):
|
||||
|
||||
- `var Commit string` — stamped at link time via
|
||||
`-ldflags "-X gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/version.Commit=<hash>"`.
|
||||
- `func Resolve() string` — returns `Commit` if non-empty; otherwise falls
|
||||
back to `debug.ReadBuildInfo()` VCS settings (`vcs.revision` truncated to
|
||||
12 chars, `-dirty` suffix when `vcs.modified=true`), so plain host builds
|
||||
(`make build`, `make run`, `go run`) are stamped for free since `.git` is
|
||||
present there; `"unknown"` when neither source is available (e.g. `go test`).
|
||||
- Internal `resolve(ldflagsCommit string, readBuildInfo func() (*debug.BuildInfo, bool)) string`
|
||||
so the fallback logic is table-testable with a fake build-info func
|
||||
(house style: stdlib testing, `t.Parallel()`, subtests,
|
||||
`Test<Function>_<scenario>` names).
|
||||
|
||||
**2. `cmd/main.go`**:
|
||||
|
||||
- Add `--version` bool flag; immediately after the existing `flag.Parse()`
|
||||
(main.go:132), if set: print `version.Resolve()` to stdout and exit 0
|
||||
(before logger/manager setup).
|
||||
- Right after `ctrl.SetLogger(...)` (main.go:134):
|
||||
`setupLog.Info("Starting egress-proxies-operator", "commit", version.Resolve(), "goVersion", runtime.Version())`
|
||||
— first line of every run, plain V(0) so it appears at any log level.
|
||||
|
||||
**3. `Makefile`**:
|
||||
|
||||
- Near the other variables:
|
||||
`GIT_COMMIT ?= $(shell git rev-parse --short=12 HEAD 2>/dev/null || echo unknown)$(shell test -z "$$(git status --porcelain 2>/dev/null)" || echo -dirty)`
|
||||
(`git status --porcelain` catches staged and untracked changes, which
|
||||
`git diff --quiet` misses).
|
||||
- `docker-build`: add `--build-arg GIT_COMMIT=$(GIT_COMMIT)`.
|
||||
- `docker-buildx`: add the same `--build-arg` to the `buildx build` line.
|
||||
- `build`/`run`/`run-dev` stay untouched — the ReadBuildInfo fallback covers them.
|
||||
|
||||
**4. `Dockerfile`**:
|
||||
|
||||
- `ARG GIT_COMMIT=unknown` in the builder stage; extend the existing
|
||||
`go build` with `-ldflags "-X <module>/internal/version.Commit=${GIT_COMMIT}"`.
|
||||
- Re-declare `ARG GIT_COMMIT` in the distroless stage and add
|
||||
`LABEL org.opencontainers.image.revision="${GIT_COMMIT}"` so the hash is
|
||||
also visible via `docker inspect` without running the binary.
|
||||
|
||||
No changes to deploy manifests, providers, or the reconciler. CHANGELOG entry
|
||||
after the user confirms it works (house convention).
|
||||
|
||||
## Verification
|
||||
|
||||
```bash
|
||||
go test ./internal/version/ # resolve() table tests
|
||||
go build ./... && go test ./... # nothing else broke
|
||||
go run ./cmd/main.go --version # host build: VCS-stamped hash (+ -dirty), exit 0
|
||||
make docker-build IMG=egress-proxies-operator:dev
|
||||
docker run --rm egress-proxies-operator:dev --version # prints the baked commit
|
||||
docker inspect egress-proxies-operator:dev \
|
||||
--format '{{index .Config.Labels "org.opencontainers.image.revision"}}'
|
||||
```
|
||||
|
||||
Live check: redeploy on the cluster and confirm the first log line carries
|
||||
`"commit"`.
|
||||
85
docs/plans/2026-08-11-1838-gcp-http-wire-logging-v5.md
Normal file
85
docs/plans/2026-08-11-1838-gcp-http-wire-logging-v5.md
Normal file
@@ -0,0 +1,85 @@
|
||||
# Plan: GCP HTTP wire logging at V(5)
|
||||
|
||||
**Created:** 2026-08-11 18:38
|
||||
|
||||
## Context
|
||||
|
||||
The GCP provider logs curated call summaries at V(1)/V(2), but when debugging
|
||||
against the real API the user wants ground truth: the actual HTTP requests and
|
||||
responses ("gory details") — visible at `--zap-log-level=5`, in the same log
|
||||
stream as everything else. The compute SDK already produces exactly this:
|
||||
`cloud.google.com/go/compute@v1.65.0/apiv1/helpers.go:60,70` logs
|
||||
`"api request"`/`"api response"` (method, URL, headers, full JSON payloads,
|
||||
lazily via `internallog.HTTPRequest/HTTPResponse`) to an injectable
|
||||
`*slog.Logger` at slog Debug level. We inject one bridged to the operator's
|
||||
zap sink, level-shifted so those Debug records surface only at V(5).
|
||||
|
||||
Level scheme after this change: V(1) call outcomes, V(2) curated detail,
|
||||
V(5) raw HTTP traffic. V(3)/V(4) reserved.
|
||||
|
||||
## Mechanism (verified in module sources)
|
||||
|
||||
- `option.WithLogger(*slog.Logger)` exists in `google.golang.org/api@v0.292.0`
|
||||
(option.go:529) and **takes precedence over `GOOGLE_SDK_GO_LOGGING_LEVEL`**
|
||||
— after this change, V(5) is the single knob for this client; document that.
|
||||
- `logr.ToSlogHandler` (go-logr/logr v1.4.3, already a direct dep) maps
|
||||
slog Debug → logr V(4), plus the base logger's V-bias. logr's own docs
|
||||
(sloghandler.go:180-184): `slog.New(ToSlogHandler(logrV2)).Debug()` ≈ V(6).
|
||||
So a base of `.V(1)` lands Debug at exactly V(5).
|
||||
- Gating is cheap: the slog handler's `Enabled()` consults the zap sink, so
|
||||
below level 5 the SDK's lazy `LogValuer`s are never evaluated.
|
||||
|
||||
## Implementation
|
||||
|
||||
**`internal/provider/gcp/gcp.go`** (only production file):
|
||||
|
||||
1. New pure function:
|
||||
```go
|
||||
// wireLogger returns the slog logger handed to the SDK: its Debug-level
|
||||
// "api request"/"api response" records (slog Debug = +4 on the logr
|
||||
// scale) land at V(5) on top of the base's V(1) shift.
|
||||
func wireLogger(base logr.Logger) *slog.Logger {
|
||||
return slog.New(logr.ToSlogHandler(base.V(1)))
|
||||
}
|
||||
```
|
||||
2. In `New` (gcp.go:96): pass it to the client —
|
||||
`compute.NewInstancesRESTClient(ctx, option.WithLogger(wireLogger(logf.Log.WithName("gcp").WithName("http"))))`.
|
||||
Base is the process-root `logf.Log` (client is built once at startup;
|
||||
`ctrl.SetLogger` runs before `registry.Build` in cmd/main.go, so it
|
||||
resolves to the real zap logger).
|
||||
3. One-time notice in `New`: if `logf.Log.V(5).Enabled()`, log at Info:
|
||||
`"GCP HTTP wire logging active — request payloads include cloud-init user-data"`
|
||||
(the secret-leak warning our curated V(2) logging exists to avoid; at V(5)
|
||||
the user has explicitly opted into raw payloads).
|
||||
4. New imports: `log/slog`, `google.golang.org/api/option` (module already in
|
||||
go.mod as a direct dep; `option` package is a first-time import in the repo).
|
||||
|
||||
**`internal/provider/gcp/gcp_test.go`**:
|
||||
|
||||
- `TestWireLogger_gatesAtV5`: table over funcr sink verbosities
|
||||
(`funcr.Options{Verbosity: N}`, pattern already used by `captureContext`):
|
||||
at 5 a `Debug("api request", ...)` through `wireLogger` emits (message and
|
||||
attrs present); at 4 it emits nothing; an `Info` record through the same
|
||||
logger lands at V(1) (sanity-check of the shift).
|
||||
- `New` itself stays untested by design (dials real Google endpoints —
|
||||
existing convention, gcp.go:86-87).
|
||||
|
||||
No changes to manifests, Makefile, other providers, or the reconciler.
|
||||
CHANGELOG entry after the user confirms it works (house convention) — this
|
||||
plus the two earlier pending entries (GCP V-logging, version stamp).
|
||||
|
||||
## Verification
|
||||
|
||||
```bash
|
||||
go test -race ./internal/provider/gcp/
|
||||
go build ./... && go test ./...
|
||||
```
|
||||
|
||||
Live (the real proof, needs the cluster):
|
||||
|
||||
```bash
|
||||
# rebuild + load image, set --zap-log-level=5, restart, then:
|
||||
kubectl -n egress-proxies-operator-system logs deploy/egress-proxies-operator-controller-manager -f \
|
||||
| grep -m2 'api request\|api response' # full URL/headers/payload visible
|
||||
# and at --zap-log-level=2: the same grep stays silent while V(2) lines still appear
|
||||
```
|
||||
135
docs/plans/2026-08-11-1935-gitea-build-workflow.md
Normal file
135
docs/plans/2026-08-11-1935-gitea-build-workflow.md
Normal file
@@ -0,0 +1,135 @@
|
||||
# Plan: Distilled Gitea Actions image-build workflow
|
||||
**Created:** 2026-08-11 19:35
|
||||
|
||||
## Context
|
||||
|
||||
This repo (`egress-proxies-operator`) has a Dockerfile, a Makefile with `docker-build`/`docker-push` targets, and a Gitea remote — but no CI workflow (CLAUDE.md flags this as a TODO). A survey of all projects under `/Users/jan.novak/srv` found 9 image-build workflows, all variations of one lineage: trigger on `workflow_dispatch` + tag push, `docker login` to `gitea.home.hrajfrisbee.cz` with `secrets.REGISTRY_TOKEN`, raw `docker build`/`docker push`, `runs-on: ubuntu-latest`, `permissions: {contents: read, packages: write}`.
|
||||
|
||||
The best individual ideas are scattered:
|
||||
- **aviso_v2**: quality-gate job before build; computes `sha-<short>` as a second immutable tag via `$GITHUB_OUTPUT`.
|
||||
- **gateway-helper-operator** (closest sibling — same kubebuilder shape): passes build args (`GIT_COMMIT` etc.), tags `:latest` alongside the version tag.
|
||||
- **psmf-data-sync test.yaml**: `actions/setup-go@v5` with `go-version-file: go.mod` + module cache (proven to work on the act_runner).
|
||||
|
||||
Goal: distill these into one `build.yaml` for this repo. User decisions: triggers = **tags + manual dispatch only** (house convention, no builds from main); build tool = **raw docker CLI** (the runner bind-mounts docker.sock, so this just works); **lightweight test gate** (`go vet` + `go build` + `go test -short`, no envtest download); **amd64 only**.
|
||||
|
||||
## The workflow
|
||||
|
||||
Create `.gitea/workflows/build.yaml`:
|
||||
|
||||
```yaml
|
||||
name: Build and Push
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
tag:
|
||||
description: 'Image tag'
|
||||
required: true
|
||||
default: 'latest'
|
||||
push:
|
||||
tags:
|
||||
- '*'
|
||||
|
||||
concurrency:
|
||||
group: build-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
check:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
cache: true
|
||||
- name: Vet
|
||||
run: go vet ./...
|
||||
- name: Build
|
||||
run: go build ./...
|
||||
- name: Test (short)
|
||||
run: go test -short ./...
|
||||
|
||||
build:
|
||||
needs: check
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Compute image tags
|
||||
id: meta
|
||||
run: |
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
TAG="${{ inputs.tag }}"
|
||||
else
|
||||
TAG="${{ github.ref_name }}"
|
||||
fi
|
||||
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
|
||||
echo "sha=sha-$(echo '${{ github.sha }}' | cut -c1-12)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Login to Gitea registry
|
||||
run: echo "${{ secrets.REGISTRY_TOKEN }}" | docker login -u ${{ github.actor }} --password-stdin gitea.home.hrajfrisbee.cz
|
||||
|
||||
- name: Build and push
|
||||
run: |
|
||||
IMAGE=gitea.home.hrajfrisbee.cz/${{ github.repository }}
|
||||
docker build \
|
||||
--build-arg GIT_COMMIT=$(echo '${{ github.sha }}' | cut -c1-12) \
|
||||
--label org.opencontainers.image.source=https://gitea.home.hrajfrisbee.cz/${{ github.repository }} \
|
||||
--label org.opencontainers.image.created=$(date -u +%Y-%m-%dT%H:%M:%SZ) \
|
||||
-t "$IMAGE:${{ steps.meta.outputs.tag }}" \
|
||||
-t "$IMAGE:${{ steps.meta.outputs.sha }}" \
|
||||
.
|
||||
docker push "$IMAGE:${{ steps.meta.outputs.tag }}"
|
||||
docker push "$IMAGE:${{ steps.meta.outputs.sha }}"
|
||||
|
||||
- name: Push latest (tag builds only)
|
||||
if: github.event_name == 'push'
|
||||
run: |
|
||||
IMAGE=gitea.home.hrajfrisbee.cz/${{ github.repository }}
|
||||
docker tag "$IMAGE:${{ steps.meta.outputs.tag }}" "$IMAGE:latest"
|
||||
docker push "$IMAGE:latest"
|
||||
```
|
||||
|
||||
### What's distilled vs. improved over the existing workflows
|
||||
|
||||
Distilled (house patterns kept as-is): triggers, `REGISTRY_TOKEN` + `github.actor` login, image name `gitea.home.hrajfrisbee.cz/${{ github.repository }}`, `ubuntu-latest`, raw docker CLI, `permissions` block.
|
||||
|
||||
Improvements none of the existing workflows have all of:
|
||||
1. **`sha-<12>` immutable tag** alongside the human tag (aviso_v2 had this; nobody else) — lets deployments pin exactly what was built.
|
||||
2. **Test gate** (aviso_v2 had one; the Go projects don't) — lightweight variant per user choice; uses `go-version-file: go.mod` so the Go version never drifts from the module.
|
||||
3. **`GIT_COMMIT` build arg** matches the Makefile/Dockerfile contract — the binary's `internal/version.Commit` and the `org.opencontainers.image.revision` label get the real commit (12-char, same width as the Makefile's `git rev-parse --short=12`; no `-dirty` needed since CI checkouts are clean).
|
||||
4. **`:latest` only on real tag pushes**, not manual dispatch — gateway-helper pushed `latest` unconditionally, which lets an ad-hoc dispatch of an old ref clobber `latest`.
|
||||
5. **`concurrency` group** — cancels a superseded run of the same ref (none of the 20 surveyed workflows have this).
|
||||
6. **OCI `source`/`created` labels** added at build time (revision label already comes from the Dockerfile).
|
||||
|
||||
## Files
|
||||
|
||||
- **Create** `.gitea/workflows/build.yaml` — content above.
|
||||
- **Update** `CLAUDE.md` — replace the `TODO: no .gitea/workflows/ CI pipeline exists yet` note in the Git Commits section with a short CI/CD subsection describing the workflow (triggers, secret, tags produced).
|
||||
- **Update** `CHANGELOG.md` — new top entry (after user confirms it works, per convention; timestamp via `date "+%Y-%m-%d %H:%M %Z"`).
|
||||
- Copy this plan to `docs/plans/YYYY-MM-DD-HHMM-gitea-build-workflow.md` (timestamp via `date "+%Y-%m-%d-%H%M"`) and commit it first, per CLAUDE.md ordering rule.
|
||||
|
||||
## Branch & MR
|
||||
|
||||
House convention: feature → own branch + MR. Dockerfile and `cmd/` already exist on `main`, so:
|
||||
|
||||
1. `git checkout -b feat/gitea-build-workflow origin/main` (do not touch the current `feat/proxy-operator` branch's uncommitted `.claude/settings.json` change — leave it be).
|
||||
2. Commit plan file, then the workflow + CLAUDE.md update (with `Co-Authored-By: Claude <noreply@anthropic.com>`).
|
||||
3. `git push -u origin feat/gitea-build-workflow`, open MR with `tea pr create --base main --head feat/gitea-build-workflow`. Do not merge.
|
||||
|
||||
Note: `main` has no `internal/`/`test/` dirs yet (those are on `feat/proxy-operator`), which is fine — the workflow only fires on tags/dispatch, and by then the operator branch will be merged. `go build ./...` / `go test -short ./...` work on both branch states.
|
||||
|
||||
## Prerequisite (user action)
|
||||
|
||||
`REGISTRY_TOKEN` secret must exist in this repo's Gitea settings (Settings → Actions → Secrets): a personal access token with `write:package` scope — same as every other project uses. Flag this in the MR description.
|
||||
|
||||
## Verification
|
||||
|
||||
The workflow doesn't trigger on branch pushes, so end-to-end verification happens after merge:
|
||||
1. Local sanity: `docker build --build-arg GIT_COMMIT=test -t scratch-check .` (confirms the build args/labels line is valid) — or at minimum a YAML parse check.
|
||||
2. After the MR merges: run the workflow manually via Gitea UI (Actions → Build and Push → Run workflow, tag `manual-test`), confirm both `manual-test` and `sha-…` tags appear under Packages, and that `:latest` was NOT updated.
|
||||
3. Then push a real version tag (e.g. `v0.1.0`) and confirm `v0.1.0`, `sha-…`, and `latest` all appear.
|
||||
58
docs/plans/2026-08-11-2152-discovery-api-docs.md
Normal file
58
docs/plans/2026-08-11-2152-discovery-api-docs.md
Normal file
@@ -0,0 +1,58 @@
|
||||
# Plan: Discovery API documentation (docs/api.md)
|
||||
**Created:** 2026-08-11 21:52
|
||||
|
||||
|
||||
## Context
|
||||
|
||||
The operator serves an HTTP discovery/lease API on `:8090` ([internal/discovery/](internal/discovery/)) that crawler clients use to list proxies, acquire TTL leases, release them, and report rate-limiting. There is no dedicated API reference today: README has four quickstart curls (no auth header, no schemas), and `docs/architecture.md` §7 has an ASCII route map. The user wants full documentation with curl examples for every feature.
|
||||
|
||||
**Decisions made with user:** doc lives in a new `docs/api.md`; commit straight to `main` (no MR).
|
||||
|
||||
## Deliverable
|
||||
|
||||
### 1. New file `docs/api.md` — full API reference
|
||||
|
||||
Content (all facts verified against code during planning):
|
||||
|
||||
- **Overview & base URL** — what the API is; in-cluster FQDN `http://egress-proxies-operator-controller-manager-discovery-service.egress-proxies-operator-system.svc.cluster.local:8090` (Service: `config/default/discovery_service.yaml`); local access via `kubectl port-forward svc/egress-proxies-operator-controller-manager-discovery-service 8090:8090`.
|
||||
- **Authentication** — static bearer token from `DISCOVERY_TOKEN` env var (populated from the optional `discovery-token` Secret, key `token`; `config/manager/manager.yaml`). Empty token ⇒ auth disabled with startup warning. Curl: `-H "Authorization: Bearer $TOKEN"` on every example. `/healthz` always exempt.
|
||||
- **Conventions** — JSON everywhere; error envelope `{"error":"<code>","message":"<text>"}`; request bodies capped at 64 KiB; proxies with a deletion timestamp are excluded from all responses.
|
||||
- **Configuration table** — `--discovery-addr` (default `:8090`), `--max-lease-ttl` (default 1h), `--lease-cooldown` (default 15m), `DISCOVERY_TOKEN`. Note the shipped Deployment passes none of these flags, so defaults apply.
|
||||
- **Endpoints**, each with request/response schema, status codes, and a copy-pasteable curl example:
|
||||
- `GET /healthz` — liveness, unauthenticated.
|
||||
- `GET /v1/proxies` — filters `healthy=true|false` (else 400 `invalid_query`) and repeatable `attr.<key>=<value>` (verbatim equality on `spec.attributes`, all pairs must match). Response `{"proxies":[proxyView...],"count":N}` sorted by id. Full `proxyView` field table: `id` (ns/name), `ip`, `port`, `attributes`, `phase` (Pending/Provisioning/Ready/Unhealthy/Deleting/Failed), `healthy` (condition `Healthy` == True), `latencyMillis`, `activeLeases`, `maxLeases` (default 5; explicit 0 = unleasable).
|
||||
- `POST /v1/leases` — body `{selector, ttlSeconds, target}` all optional; TTL defaults 5m, capped at max-lease-ttl (else 400 `invalid_ttl`). 201 `{leaseID, proxy, expiresAt, ttlSeconds}`; 409 `no_match` with `considered/atCapacity/inCooldown/unhealthy` counts (documented meanings).
|
||||
- `DELETE /v1/leases/{id}` — early release; always 204, idempotent.
|
||||
- `POST /v1/leases/{id}/report` — body `{result: ok|rate_limited|banned, target}`; 204, 404 `unknown_lease`, 400 `invalid_result`. `ok` is a pure ack; `rate_limited` and `banned` behave identically today (both start one cooldown window).
|
||||
- **Selection & cooldown semantics** (short section — this is the non-obvious part clients need):
|
||||
- Selection order: fewest active leases → lowest latency → lexicographic id; deterministic; only healthy, non-deleting proxies with free capacity are candidates.
|
||||
- Cooldown: 15m default (`--lease-cooldown`), keyed `{proxy, target}`. Report with a target blocks only leases requesting that target; report without a target (and lease without one) creates a **global** cooldown blocking all acquisitions of that proxy — call this footgun out explicitly.
|
||||
- Expired leases stay reportable for one cooldown window past TTL.
|
||||
- **End-to-end workflow example** — numbered curl walkthrough: acquire → use `proxy.ip:port` as HTTP proxy (`curl -x`) → report `rate_limited` on 429 → release. Using a `jq`-extracted `leaseID`.
|
||||
- **Caveats** — lease/cooldown state is in-memory and per-process: single replica only, operator restart drops all leases and cooldowns.
|
||||
|
||||
### 2. Pointers to the new doc (small edits)
|
||||
|
||||
- `README.md`: one-line link near the quickstart curl section ("full reference: docs/api.md").
|
||||
- `docs/architecture.md` §7: one-line link to `docs/api.md` as the detailed reference.
|
||||
|
||||
### 3. Housekeeping per CLAUDE.md
|
||||
|
||||
- First action post-approval: copy this plan to `docs/plans/<timestamp>-discovery-api-docs.md` (timestamp from `date "+%Y-%m-%d-%H%M"`), commit it alone.
|
||||
- Then write the docs, commit to `main` with `Co-Authored-By: Claude <noreply@anthropic.com>` trailer, push.
|
||||
- Append execution summary + status checklist to `docs/plans-executions/<same-timestamp>-discovery-api-docs.md` in the docs commit.
|
||||
- Add `CHANGELOG.md` entry (timestamp via `date "+%Y-%m-%d %H:%M %Z"`) once the user confirms.
|
||||
|
||||
## Key source files (facts source of truth)
|
||||
|
||||
- [internal/discovery/handlers.go](internal/discovery/handlers.go), [internal/discovery/server.go](internal/discovery/server.go) — routes, schemas, status codes, auth, limits.
|
||||
- [internal/lease/store.go](internal/lease/store.go) — selection order, cooldown/retention, stats.
|
||||
- [api/v1alpha1/proxy_types.go](api/v1alpha1/proxy_types.go), [api/v1alpha1/helpers.go](api/v1alpha1/helpers.go) — defaults (port 3128, maxLeases 5), phases, conditions.
|
||||
- [cmd/main.go](cmd/main.go) — flags/env defaults.
|
||||
- [config/default/discovery_service.yaml](config/default/discovery_service.yaml), [config/manager/manager.yaml](config/manager/manager.yaml) — service DNS, token secret.
|
||||
|
||||
## Verification
|
||||
|
||||
- Cross-check every documented status code / field name against `internal/discovery/server_test.go` expectations.
|
||||
- Sanity-run `go build ./... && go test -short ./internal/discovery/ ./internal/lease/` (no code changes expected — confirms docs match current behavior, not a stale tree).
|
||||
- Optionally lint the curl JSON bodies by piping each through `jq .` locally.
|
||||
33
docs/plans/2026-08-11-2220-demo-scripts.md
Normal file
33
docs/plans/2026-08-11-2220-demo-scripts.md
Normal file
@@ -0,0 +1,33 @@
|
||||
# Plan: Demo script — egress IP check through each healthy proxy
|
||||
**Created:** 2026-08-11 22:20
|
||||
|
||||
|
||||
## Context
|
||||
|
||||
First of a planned series of demo scripts under `docs/demo/`. This one showcases the discovery API end to end without leases: list proxies from `$BASE_URL/v1/proxies`, and for each **healthy** one, call an IP-echo site through it (`curl -x`) to show the egress IP that proxy provides — clearly labeling which proxy each request goes through. The user explicitly asked to start by switching to a new branch. The script will be iterated on: **write it but do not commit it** — the user wants to add things to it before anything is committed.
|
||||
|
||||
## Steps
|
||||
|
||||
### Step 0 — Branch
|
||||
|
||||
Create `feat/demo-scripts` off `main`. Commit only the plan copy (`docs/plans/<timestamp>-demo-scripts.md`, timestamp via `date "+%Y-%m-%d-%H%M"`) per CLAUDE.md — nothing else gets committed this round.
|
||||
|
||||
### Step 1 — `docs/demo/show-egress-ips.sh` (new file, executable, left uncommitted)
|
||||
|
||||
Style: match [.devcontainer/post-install.sh](.devcontainer/post-install.sh) — `#!/bin/bash`, `set -euo pipefail`, `ERROR:`/`WARNING:` messages, `${VAR}` braces.
|
||||
|
||||
Behavior:
|
||||
|
||||
1. **Args/env:** `BASE_URL` is `$1` (required; missing → usage text + exit 1, e.g. `usage: show-egress-ips.sh <BASE_URL> (e.g. localhost:8090)`). Optional env: `TOKEN` (bearer token, same name docs/api.md uses; when set, send `Authorization: Bearer $TOKEN`), `IP_ECHO_URL` (default `https://api.ipify.org?format=json` — returns `{"ip":"..."}`).
|
||||
2. **Dependency check:** `command -v curl`, `command -v jq` → `ERROR` + exit 1 if missing.
|
||||
3. **Fetch** `"$BASE_URL/v1/proxies"` once (no server-side `healthy` filter — fetch all so unhealthy ones can be shown as skipped, which makes the demo more informative). Fail with a clear error if curl or JSON parsing fails.
|
||||
4. **Iterate** proxies with `jq -c '.proxies[]'`; for each, extract `id`, `ip`, `port`, `healthy`:
|
||||
- unhealthy → print `--- skipping <id> (unhealthy) ---`
|
||||
- healthy → print a clear banner naming the proxy before the request, e.g. `=== via <id> — http://<ip>:<port> ===`, then `curl -sS --max-time 10 -x "http://${ip}:${port}" "$IP_ECHO_URL"`; print the JSON response. A failed probe prints `WARNING: request through <id> failed` and continues (guard so `set -e` doesn't kill the loop).
|
||||
5. Finish with a one-line summary: N proxies, M probed, K skipped/failed.
|
||||
|
||||
Reference for API shapes: [docs/api.md](docs/api.md) (`proxies[].id/ip/port/healthy`; `curl -x http://ip:port` usage is already documented there and in README).
|
||||
|
||||
### Deliberately deferred (user will iterate on the script first)
|
||||
|
||||
- No commit of the script, no push beyond the plan commit, no MR, no execution summary, no CHANGELOG — all wait until the user says the script (or script set) is ready.
|
||||
131
docs/reviews/2026-08-10-1134-proxy-operator-pr-review.md
Normal file
131
docs/reviews/2026-08-10-1134-proxy-operator-pr-review.md
Normal file
@@ -0,0 +1,131 @@
|
||||
# PR review findings: feat/proxy-operator
|
||||
|
||||
**Created:** 2026-08-10 11:34
|
||||
**Scope:** `origin/main...feat/proxy-operator` (merge-base 076bc66, 25 commits, ~80 files)
|
||||
**Reviewers:** `go-operator-reviewer` + `operator-reviewer` agents; findings consolidated, most severe first. Check off items as they're processed.
|
||||
|
||||
Both reviewers rated the core reconcile architecture sound: single status writer with one
|
||||
deferred patch, finalizer added before any provider call, Get-before-RemoveFinalizer on
|
||||
delete, CEL immutability rules correctly split to avoid the oldSelf-on-CREATE trap,
|
||||
leader-election gating on destructive runnables, GC tombstone rules (MinAge, UID-less
|
||||
instances never deleted).
|
||||
|
||||
## Merge-blockers
|
||||
|
||||
- [ ] **Discovery leases proxies with an empty IP** — found independently by both reviewers.
|
||||
`internal/discovery/handlers.go:151`, `internal/controller/proxy_controller.go:226`
|
||||
During instance replacement (and the Get→NotFound recovery path) the reconciler clears
|
||||
`status.ip` but only the create branch removes the `Healthy` condition, and the health
|
||||
engine prunes state for empty-host proxies so nothing refreshes it. For the whole
|
||||
delete→recreate window (minutes on GCP), `isHealthy` still returns true and
|
||||
`handleAcquireLease` grants `201 Created` with `"ip": ""`, burning a `MaxLeases` slot.
|
||||
**Fix:** add `EffectiveHost() != ""` to `isHealthy` (covers list + acquire), and
|
||||
remove/downgrade `Healthy` wherever `status.IP` is cleared.
|
||||
|
||||
- [ ] **Orphan GC deletes other installations' fleets in a shared GCP project.**
|
||||
`internal/provider/gcp/insert.go:57`, `internal/gc/gc.go:93`
|
||||
Instances are tagged only `proxy-operator-managed=true` + CR UID; the sweeper deletes any
|
||||
tagged instance whose UID isn't in *its own cluster's* Proxy list. Two clusters sharing a
|
||||
GCP project delete each other's VMs every GC interval in a permanent loop.
|
||||
**Fix:** add an installation-identity label (cluster/deployment ID) set by both providers
|
||||
and filtered on in `ListByTag`.
|
||||
|
||||
- [ ] **Permanent-error latch wedges proxies on failures that aren't spec-caused.**
|
||||
`internal/controller/proxy_controller.go:126`
|
||||
Latch keys on `observedGeneration == generation`, but two failure inputs live outside the
|
||||
spec: an unconfigured provider (config fix + restart doesn't bump generation, and
|
||||
`spec.provider` is CEL-immutable → stuck `Failed` short of deleting the CR) and resolved
|
||||
Secret content (Secret fix enqueues a reconcile that short-circuits at the latch before
|
||||
re-resolving cloud-init).
|
||||
**Fix:** latch should also consider current spec-hash / provider availability.
|
||||
|
||||
## Worth fixing
|
||||
|
||||
- [ ] **Deletion-path failures invisible in status** — flagged by both reviewers.
|
||||
`internal/controller/proxy_controller.go:319`, `:266`
|
||||
`deletionFailure` swallows `ErrQuotaExceeded` (nil error, no status write); unconfigured
|
||||
provider returns a bare error forever. A Proxy wedged in `Deleting` shows nothing in
|
||||
`kubectl describe`. Stage `setProvisioned(p, False, ReasonDeleting, ...)` before returning.
|
||||
Also: `Delete` is resubmitted on every `DeletionPoll` pass, churning GCP quota — a state
|
||||
check on the `Get` result would avoid it.
|
||||
|
||||
- [ ] **Lost providerID on `setSpecHash` conflict.**
|
||||
`internal/controller/proxy_controller.go:161`
|
||||
On Update conflict the function returns before `p.Status.ProviderID = id`, so the deferred
|
||||
patch persists an empty providerID for a just-created instance. Self-heals via GC.
|
||||
**Fix:** set `p.Status.ProviderID = id` before returning the error (one line).
|
||||
|
||||
- [ ] **Terminating pods still report `StateRunning`.**
|
||||
`internal/provider/kubernetes/kubernetes.go:151`
|
||||
A pod with a deletionTimestamp keeps `phase=Running` + `PodIP` while terminating, so drift
|
||||
reconcile republishes `Provisioned=True` and discovery keeps leasing a dying pod.
|
||||
**Fix:** map non-zero `pod.DeletionTimestamp` to `StateTerminated` in `instanceFromPod`.
|
||||
|
||||
- [ ] **Stale-cache spec-hash race deletes the freshly created replacement instance.**
|
||||
`internal/controller/proxy_controller.go:176`
|
||||
Instance name derives from CR UID, so old and new instances share a providerID. A reconcile
|
||||
served a cached object from before a just-completed replacement re-enters `replaceInstance`
|
||||
and deletes the *new* healthy instance. Converges, but destroys a good instance.
|
||||
**Fix:** re-read uncached before the destructive branch, or compare `inst.CreatedAt`
|
||||
against the annotation-update time.
|
||||
|
||||
- [ ] **`observedGeneration` written before the generation is actually processed.**
|
||||
`internal/controller/status.go:114`
|
||||
Set unconditionally in `patchStatusIfChanged`, including on the finalizer-add pass and
|
||||
`resolveCloudInit` failures — misleads kstatus-style tooling. Set it only once the state
|
||||
machine has genuinely evaluated the spec.
|
||||
|
||||
- [ ] **No event filtering on the Proxy watch.**
|
||||
`internal/controller/proxy_controller.go:402`
|
||||
Every self-inflicted status patch triggers a follow-up reconcile with an extra cloud `Get`,
|
||||
roughly doubling provider read traffic. Caution: a plain `GenerationChangedPredicate`
|
||||
breaks the finalizer flow (relies on its own Update event to re-enter) — needs a
|
||||
status-only/resourceVersion-only filter or an explicit requeue in the finalizer pass.
|
||||
|
||||
- [ ] **Unlabelled cloud-init Secrets produce a misleading NotFound with endless backoff.**
|
||||
`internal/controller/proxy_controller.go:344`, `cmd/main.go:210`
|
||||
The label-restricted cache turns "exists but missing `crawl.example.com/cloud-init=true`"
|
||||
into `CloudInitError: not found`. Mention the label requirement in the condition message,
|
||||
or read via uncached `APIReader` and validate the label explicitly.
|
||||
|
||||
- [ ] **RBAC over-grant.**
|
||||
`config/rbac/role.yaml:25`
|
||||
`create;delete` on `proxies` is scaffold residue (controller never creates/deletes CRs);
|
||||
cluster-wide `pods create/delete` and `secrets get/list/watch` apply even when only the GCP
|
||||
provider is configured — pod rules belong in an optional kustomize component.
|
||||
|
||||
## Simplifications
|
||||
|
||||
- [ ] **Delete `internal/provider/registry`** — 14 lines of logic, one caller
|
||||
(`cmd/main.go:148`); fold `Build`/`Constructor` into the composition root. Also fixes the
|
||||
two-sources-of-truth problem: `internal/provider/config.go:84` hardcodes
|
||||
`"kubernetes"`/`"gcp"` while `registry.Build` dispatches through a caller-supplied map —
|
||||
validate against the constructor map instead. Net −1 package, −44 lines, −92 test lines.
|
||||
|
||||
- [ ] **Collapse `LeaseStore` interface to `*lease.Store`.**
|
||||
`internal/discovery/server.go:28`
|
||||
Single implementation and not a test seam (tests wire the real `lease.NewStore`).
|
||||
Keep `HealthSnapshotter` and `instancesAPI` — those are genuine seams.
|
||||
|
||||
- [ ] **Replace metrics nil-guards with no-op defaults.**
|
||||
`internal/health/engine.go:68`, `internal/discovery/server.go:38`,
|
||||
`internal/provider/metrics.go:8`
|
||||
Keep the interfaces (legit "no prometheus in domain packages" rationale) but default the
|
||||
fields to a no-op impl — `provider.WithMetrics` already dereferences unconditionally, so
|
||||
the guards are inconsistent anyway.
|
||||
|
||||
## Nice-to-have
|
||||
|
||||
- [ ] Add an `OwnerReference` to provider pods (`internal/provider/kubernetes/pod.go:24`) —
|
||||
free cascading deletion if the finalizer is ever bypassed; `CreateRequest` already carries
|
||||
Namespace/ProxyName/UID.
|
||||
- [ ] `Close()` the GCP `*compute.InstancesClient` (`internal/provider/gcp/gcp.go:89`) —
|
||||
harmless today, a leak the moment providers are rebuilt on config reload.
|
||||
- [ ] Fix `.golangci` config: it references a missing `logcheck` plugin, so the linter only
|
||||
runs with the project config disabled.
|
||||
- [ ] External-mode endpoint edits don't reset health-engine counters
|
||||
(`internal/health/engine.go:206` keys on name+UID): flipping `endpoint.host` keeps the old
|
||||
host's `Healthy=True` for `failureThreshold × interval`. Arguably a replacement, not a flap.
|
||||
- [ ] `init()` funcs at `api/v1alpha1/proxy_types.go:353` and `cmd/main.go:67` conflict with
|
||||
the repo's "no `init()`" convention; kubebuilder-idiomatic, but scheme registration could
|
||||
use the scaffold's `SchemeBuilder.Register` at package var scope.
|
||||
123
docs/testing.md
Normal file
123
docs/testing.md
Normal file
@@ -0,0 +1,123 @@
|
||||
# Testing
|
||||
|
||||
What tests exist, how to run them, and where the boundaries of the
|
||||
automated suites are. The philosophy throughout: **test against the most
|
||||
real double available** — a real envtest API server over a fake client, a
|
||||
real CONNECT-capable proxy stub over a mocked HTTP client, the real lease
|
||||
store under the discovery handlers — and leave the gaps that only real
|
||||
infrastructure can close to the kind verification run, documented at the
|
||||
bottom.
|
||||
|
||||
## Running
|
||||
|
||||
```sh
|
||||
make test # THE canonical run: codegen + fmt + vet + envtest setup,
|
||||
# then `go test -race` on every package, with coverage
|
||||
go test -short ./... # skip the envtest suite (runs in <1s per package)
|
||||
go tool cover -html=cover.out # browse coverage from the last make test
|
||||
```
|
||||
|
||||
Targeted runs:
|
||||
|
||||
```sh
|
||||
go test -race ./internal/lease/ # one package
|
||||
go test -race -run TestAcquire ./internal/lease/ # one test (or a prefix)
|
||||
go test -race -count=2 ./internal/health/ # flake-shaking: run twice
|
||||
```
|
||||
|
||||
**Gotcha — plain `go test ./...` fails the controller package** with an
|
||||
error like `/usr/local/kubebuilder/bin/etcd: no such file`. That is not a
|
||||
code problem: the envtest suite needs `KUBEBUILDER_ASSETS` pointing at the
|
||||
API-server/etcd binaries, which only `make test` sets up (via
|
||||
`setup-envtest`). Either use `make test`, or export it once:
|
||||
|
||||
```sh
|
||||
KUBEBUILDER_ASSETS="$PWD/bin/k8s/<version>-<os>-<arch>" go test -race ./...
|
||||
```
|
||||
|
||||
Everything runs with `-race` — `make test` enforces it; keep the flag on
|
||||
manual runs too.
|
||||
|
||||
## The suites, package by package
|
||||
|
||||
| Package | Files | What is covered | Test double / technique |
|
||||
|---|---|---|---|
|
||||
| `api/v1alpha1` | `helpers_test.go` | `EffectivePort/EffectiveHost/HealthCheckOrDefault/MaxLeasesOrDefault` | pure table-driven; CEL rules are **not** unit-testable — see envtest below |
|
||||
| `internal/controller` | `reconcile_test.go` | every row of the reconciler's action table: create, poll, publish IP, replace, adopt, NotFound recovery, delete/finalizer, quota/permanent/transient errors, cloud-init resolution | `controller-runtime/pkg/client/fake` + an in-test `stubProvider` |
|
||||
| | `status_test.go` | `computePhase` truth table (10 rows) | pure |
|
||||
| | `spechash_test.go` | hash stability, nil/empty normalization, sensitivity to every replacement-triggering field | pure |
|
||||
| | `health_test.go` | Healthy-condition representation: Ready/Unhealthy phases, no-verdict, stale-verdict cleanup on replacement, nil snapshotter | fake client + `fakeSnapshotter` |
|
||||
| | `proxy_controller_test.go` (envtest, ginkgo) | full lifecycles against a **real API server**: provision→Running, spec-change replacement, finalizer deletion, External tracking, Ready-through-health, quota-vs-permanent, adoption; **all CEL rules** (six invalid creates, mode/provider immutability incl. removal) and the `default={}` materialization | envtest apiserver + stub provider; the only place CEL and structural defaulting actually execute |
|
||||
| `internal/provider` | `name_test.go` | deterministic naming: idempotency, `^proxy-[a-z2-7]{16}$`, 10k-UID distinctness | pure |
|
||||
| | `errors_test.go` | taxonomy: `Class()` mapping, `errors.Is` **and** `errors.As` through the multi-unwrap | pure |
|
||||
| | `config_test.go` | providers-config parsing + fail-fast validation | pure |
|
||||
| | `metrics_test.go` | `WithMetrics` decorator: classified result labels, error passthrough | fake recorder + static provider |
|
||||
| `internal/provider/kubernetes` | `kubernetes_test.go`, `pod_test.go` | Create/Get/Delete/ListByTag over real Pod objects; pure `buildPod`/`squidConf` incl. the `max_filedescriptors` OOM-regression assertion | `client/fake` (real `corev1.Pod`s, no kubelet) |
|
||||
| `internal/provider/gcp` | `insert_test.go`, `errors_test.go`, `gcp_test.go` | field-by-field `buildInsertRequest`; HTTP-code classification table; zone-qualified IDs, 409-is-success, 9-row state mapping, ListByTag filter + `ReturnPartialSuccess` | pure builder needs **no fake**; the rest uses the flattened `instancesAPI` seam |
|
||||
| `internal/health` | `probe_test.go` | probes through a **real CONNECT-capable proxy stub** to a real TLS server: tunnel success, refused CONNECT, wrong status, dead proxy, plain-http forward | `httptest` + hijacked bidirectional tunnel |
|
||||
| | `engine_test.go` | thresholds, first-verdict, latency suppression matrix, dropped-event retry, stale-probe UID guard, tick scheduling/seeding/pruning, `Start` end-to-end | fake `client.Reader` + injected `probeFn` |
|
||||
| `internal/lease` | `store_test.go` | capacity, `MaxLeases=0`, all three selection tie-breaks, target-scoped vs global cooldowns, expiry via fake clock, report-on-retained-lease, idempotent release, **40 concurrent acquires never exceeding capacity** | injectable clock; the concurrency case is why `-race` matters |
|
||||
| `internal/discovery` | `server_test.go` | auth matrix (incl. `/healthz` exemption), list filtering, grant shape, full 409 arithmetic, invalid TTL/body/result, idempotent release, report→cooldown→409 round trip, `Start`/shutdown | `httptest` over the real handler chain, fake cache reader, **real** `lease.Store` |
|
||||
| `internal/gc` | `gc_test.go` | the true-orphan matrix (live/deleting/young/unlabelled all kept), broken-provider isolation, list-failure skips sweep, namespace guard, sweep loop | fake reader + canned-list provider |
|
||||
| `internal/metrics` | `metrics_test.go` | scrape-time collectors (`GatherAndCompare`), per-proxy series cleanup via `ForgetProxy`, label counts, fresh-registry-per-test property | `prometheus/client_golang/testutil` |
|
||||
|
||||
Conventions (from `CLAUDE.md`): stdlib `testing`, table-driven by default,
|
||||
`t.Parallel()` where safe, tests next to source, ginkgo only in the
|
||||
envtest suite the scaffold generated.
|
||||
|
||||
## What the automated suites deliberately do NOT cover
|
||||
|
||||
- **`New()` constructors that dial real infrastructure**: the kubernetes
|
||||
provider's `New` (connects to whatever your kubeconfig points at), the
|
||||
GCP provider's `New` and its `realInstances` SDK adapter (ADC + real
|
||||
Google endpoints). Both are thin; both are exercised by the kind run /
|
||||
real deployments. Their 0% coverage is by design — do not "fix" it.
|
||||
- **CEL and structural defaulting under the fake client**: the fake
|
||||
client runs neither. Every unit test that relies on defaults calls the
|
||||
`*OrDefault` helpers; every CEL rule is asserted in the envtest suite
|
||||
instead.
|
||||
- **A container actually starting and serving**: envtest has no kubelet,
|
||||
so a Pod created there sits Pending forever. Whether Squid really comes
|
||||
up and tunnels CONNECT is provable only on a real cluster — that is the
|
||||
kind verification's job, and it is exactly what caught the Squid OOM
|
||||
bug (below).
|
||||
- **Live GCP**: no test talks to Google. The seam boundary
|
||||
(`buildInsertRequest` + classification) is tested exhaustively instead.
|
||||
|
||||
## The scaffolded `make test-e2e` suite
|
||||
|
||||
`test/e2e/` is the kubebuilder-generated smoke suite (build image → kind
|
||||
cluster → deploy → assert the manager pod runs and serves metrics). It
|
||||
compiles only under `-tags=e2e`, manages its own kind cluster
|
||||
(`make test-e2e` / `make cleanup-test-e2e`), and has been kept compiling
|
||||
(`go vet -tags=e2e ./...` is part of the routine) but is **not part of
|
||||
`make test` and was not used for the release verification** — the manual
|
||||
kind run below covers strictly more. Treat it as scaffold to grow into if
|
||||
CI wants an automated cluster smoke test.
|
||||
|
||||
## The kind verification run (the real end-to-end)
|
||||
|
||||
The README quickstart **is** the e2e test, run by hand before the MR:
|
||||
deploy the operator in-cluster on a throwaway kind cluster, watch a real
|
||||
Squid pod reach `Ready` via a real CONNECT probe, then exercise the whole
|
||||
lease lifecycle through the discovery API (list → lease → report
|
||||
`rate_limited` → same-target lease answered 409 `inCooldown:1` → release
|
||||
204 twice) and delete the CR to watch the finalizer remove the pod.
|
||||
|
||||
Worth knowing about that run (full detail in
|
||||
[plans-executions/2026-08-07-1747-proxy-operator.md](plans-executions/2026-08-07-1747-proxy-operator.md),
|
||||
"Verification" section): it was not a clean pass-through. It caught two
|
||||
real bugs the automated suites structurally could not —
|
||||
|
||||
1. `make run-dev` on a laptop can never produce a `Ready`
|
||||
kubernetes-provider proxy: health probes originate on the host, which
|
||||
cannot route to kind's pod network. The quickstart was rewritten to
|
||||
deploy in-cluster.
|
||||
2. Squid was OOM-killed at startup: it sizes file-descriptor tables from
|
||||
`RLIMIT_NOFILE`, which containerd under kind sets effectively
|
||||
unlimited. Fixed with `max_filedescriptors 1024` in the generated
|
||||
config, plus a regression assertion in `pod_test.go`.
|
||||
|
||||
That is the pattern to keep: when the kind run finds something, the fix
|
||||
lands **with a unit-level regression test**, so the manual run stays a
|
||||
discovery tool rather than a recurring gate.
|
||||
57
go.mod
57
go.mod
@@ -3,15 +3,25 @@ module gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator
|
||||
go 1.26.0
|
||||
|
||||
require (
|
||||
cloud.google.com/go/compute v1.65.0
|
||||
github.com/go-logr/logr v1.4.3
|
||||
github.com/onsi/ginkgo/v2 v2.27.4
|
||||
github.com/onsi/gomega v1.39.0
|
||||
github.com/prometheus/client_golang v1.23.2
|
||||
google.golang.org/api v0.292.0
|
||||
google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af
|
||||
k8s.io/api v0.36.0
|
||||
k8s.io/apimachinery v0.36.0
|
||||
k8s.io/client-go v0.36.0
|
||||
sigs.k8s.io/controller-runtime v0.24.1
|
||||
sigs.k8s.io/yaml v1.6.0
|
||||
)
|
||||
|
||||
require (
|
||||
cel.dev/expr v0.25.1 // indirect
|
||||
cel.dev/expr v0.25.2 // indirect
|
||||
cloud.google.com/go/auth v0.22.0 // indirect
|
||||
cloud.google.com/go/auth/oauth2adapt v0.2.8 // indirect
|
||||
cloud.google.com/go/compute/metadata v0.9.0 // indirect
|
||||
github.com/Masterminds/semver/v3 v3.4.0 // indirect
|
||||
github.com/antlr4-go/antlr/v4 v4.13.0 // indirect
|
||||
github.com/beorn7/perks v1.0.1 // indirect
|
||||
@@ -24,7 +34,6 @@ require (
|
||||
github.com/felixge/httpsnoop v1.0.4 // indirect
|
||||
github.com/fsnotify/fsnotify v1.9.0 // indirect
|
||||
github.com/fxamacker/cbor/v2 v2.9.0 // indirect
|
||||
github.com/go-logr/logr v1.4.3 // indirect
|
||||
github.com/go-logr/stdr v1.2.2 // indirect
|
||||
github.com/go-logr/zapr v1.3.0 // indirect
|
||||
github.com/go-openapi/jsonpointer v0.21.0 // indirect
|
||||
@@ -35,17 +44,20 @@ require (
|
||||
github.com/google/gnostic-models v0.7.0 // indirect
|
||||
github.com/google/go-cmp v0.7.0 // indirect
|
||||
github.com/google/pprof v0.0.0-20250403155104-27863c87afa6 // indirect
|
||||
github.com/google/s2a-go v0.1.9 // indirect
|
||||
github.com/google/uuid v1.6.0 // indirect
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.3.19 // indirect
|
||||
github.com/googleapis/gax-go/v2 v2.23.0 // indirect
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.27.7 // indirect
|
||||
github.com/inconshreveable/mousetrap v1.1.0 // indirect
|
||||
github.com/josharian/intern v1.0.0 // indirect
|
||||
github.com/json-iterator/go v1.1.12 // indirect
|
||||
github.com/kylelemons/godebug v1.1.0 // indirect
|
||||
github.com/mailru/easyjson v0.7.7 // indirect
|
||||
github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect
|
||||
github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee // indirect
|
||||
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect
|
||||
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 // indirect
|
||||
github.com/prometheus/client_golang v1.23.2 // indirect
|
||||
github.com/prometheus/client_model v0.6.2 // indirect
|
||||
github.com/prometheus/common v0.67.5 // indirect
|
||||
github.com/prometheus/procfs v0.19.2 // indirect
|
||||
@@ -54,37 +66,37 @@ require (
|
||||
github.com/stoewer/go-strcase v1.3.0 // indirect
|
||||
github.com/x448/float16 v0.8.4 // indirect
|
||||
go.opentelemetry.io/auto/sdk v1.2.1 // indirect
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.65.0 // indirect
|
||||
go.opentelemetry.io/otel v1.41.0 // indirect
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.67.0 // indirect
|
||||
go.opentelemetry.io/otel v1.44.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.40.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.40.0 // indirect
|
||||
go.opentelemetry.io/otel/metric v1.41.0 // indirect
|
||||
go.opentelemetry.io/otel/sdk v1.40.0 // indirect
|
||||
go.opentelemetry.io/otel/trace v1.41.0 // indirect
|
||||
go.opentelemetry.io/otel/metric v1.44.0 // indirect
|
||||
go.opentelemetry.io/otel/sdk v1.44.0 // indirect
|
||||
go.opentelemetry.io/otel/trace v1.44.0 // indirect
|
||||
go.opentelemetry.io/proto/otlp v1.9.0 // indirect
|
||||
go.uber.org/multierr v1.11.0 // indirect
|
||||
go.uber.org/zap v1.27.1 // indirect
|
||||
go.yaml.in/yaml/v2 v2.4.3 // indirect
|
||||
go.yaml.in/yaml/v3 v3.0.4 // indirect
|
||||
golang.org/x/crypto v0.54.0 // indirect
|
||||
golang.org/x/exp v0.0.0-20251219203646-944ab1f22d93 // indirect
|
||||
golang.org/x/mod v0.32.0 // indirect
|
||||
golang.org/x/net v0.49.0 // indirect
|
||||
golang.org/x/oauth2 v0.34.0 // indirect
|
||||
golang.org/x/sync v0.19.0 // indirect
|
||||
golang.org/x/sys v0.40.0 // indirect
|
||||
golang.org/x/term v0.39.0 // indirect
|
||||
golang.org/x/text v0.33.0 // indirect
|
||||
golang.org/x/time v0.14.0 // indirect
|
||||
golang.org/x/tools v0.41.0 // indirect
|
||||
golang.org/x/mod v0.37.0 // indirect
|
||||
golang.org/x/net v0.57.0 // indirect
|
||||
golang.org/x/oauth2 v0.36.0 // indirect
|
||||
golang.org/x/sync v0.22.0 // indirect
|
||||
golang.org/x/sys v0.47.0 // indirect
|
||||
golang.org/x/term v0.45.0 // indirect
|
||||
golang.org/x/text v0.40.0 // indirect
|
||||
golang.org/x/time v0.15.0 // indirect
|
||||
golang.org/x/tools v0.47.0 // indirect
|
||||
gomodules.xyz/jsonpatch/v2 v2.4.0 // indirect
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260128011058-8636f8732409 // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260128011058-8636f8732409 // indirect
|
||||
google.golang.org/grpc v1.79.3 // indirect
|
||||
google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af // indirect
|
||||
google.golang.org/genproto v0.0.0-20260319201613-d00831a3d3e7 // indirect
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260630182238-925bb5da69e7 // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260803160001-6ac0973c030d // indirect
|
||||
google.golang.org/grpc v1.83.0 // indirect
|
||||
gopkg.in/evanphx/json-patch.v4 v4.13.0 // indirect
|
||||
gopkg.in/inf.v0 v0.9.1 // indirect
|
||||
gopkg.in/yaml.v3 v3.0.1 // indirect
|
||||
k8s.io/api v0.36.0 // indirect
|
||||
k8s.io/apiextensions-apiserver v0.36.0 // indirect
|
||||
k8s.io/apiserver v0.36.0 // indirect
|
||||
k8s.io/component-base v0.36.0 // indirect
|
||||
@@ -96,5 +108,4 @@ require (
|
||||
sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 // indirect
|
||||
sigs.k8s.io/randfill v1.0.0 // indirect
|
||||
sigs.k8s.io/structured-merge-diff/v6 v6.3.2 // indirect
|
||||
sigs.k8s.io/yaml v1.6.0 // indirect
|
||||
)
|
||||
|
||||
102
go.sum
102
go.sum
@@ -1,5 +1,15 @@
|
||||
cel.dev/expr v0.25.1 h1:1KrZg61W6TWSxuNZ37Xy49ps13NUovb66QLprthtwi4=
|
||||
cel.dev/expr v0.25.1/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4=
|
||||
cel.dev/expr v0.25.2 h1:K6j46C81hXtZQfuX60cVWQFBJahKSE2gfRbNuvr5bFs=
|
||||
cel.dev/expr v0.25.2/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4=
|
||||
cloud.google.com/go v0.123.0 h1:2NAUJwPR47q+E35uaJeYoNhuNEM9kM8SjgRgdeOJUSE=
|
||||
cloud.google.com/go v0.123.0/go.mod h1:xBoMV08QcqUGuPW65Qfm1o9Y4zKZBpGS+7bImXLTAZU=
|
||||
cloud.google.com/go/auth v0.22.0 h1:Xp9wAKkLoeaYb5pYZZoQGz4E9sdPxIbzS3gywZE3ciQ=
|
||||
cloud.google.com/go/auth v0.22.0/go.mod h1:M9o2Oz+YI2jAfxewJgb1vyI3vceHF+eohmxyzmrl+9s=
|
||||
cloud.google.com/go/auth/oauth2adapt v0.2.8 h1:keo8NaayQZ6wimpNSmW5OPc283g65QNIiLpZnkHRbnc=
|
||||
cloud.google.com/go/auth/oauth2adapt v0.2.8/go.mod h1:XQ9y31RkqZCcwJWNSx2Xvric3RrU88hAYYbjDWYDL+c=
|
||||
cloud.google.com/go/compute v1.65.0 h1:K0a3NRvazE7sZn5qswwI6BtlaZv1fgR5wFop5LZCLz8=
|
||||
cloud.google.com/go/compute v1.65.0/go.mod h1:vFq+Ztj9Rzhc8zf1t6hGp/6NdrEVG1GakkyVRQPRgKc=
|
||||
cloud.google.com/go/compute/metadata v0.9.0 h1:pDUj4QMoPejqq20dK0Pg2N4yG9zIkYGdBtwLoEkH9Zs=
|
||||
cloud.google.com/go/compute/metadata v0.9.0/go.mod h1:E0bWwX5wTnLPedCKqk3pJmVgCBSM6qQI1yTBdEb3C10=
|
||||
github.com/Masterminds/semver/v3 v3.4.0 h1:Zog+i5UMtVoCU8oKka5P7i9q9HgrJeGzI9SA1Xbatp0=
|
||||
github.com/Masterminds/semver/v3 v3.4.0/go.mod h1:4V+yj/TJE1HU9XfppCwVMZq3I84lprf4nC11bSS5beM=
|
||||
github.com/antlr4-go/antlr/v4 v4.13.0 h1:lxCg3LAv+EUK6t1i0y1V6/SLeUi0eKEKdhQAlS8TVTI=
|
||||
@@ -68,8 +78,14 @@ github.com/google/gofuzz v1.2.0 h1:xRy4A+RhZaiKjJ1bPfwQ8sedCA+YS2YcCHW6ec7JMi0=
|
||||
github.com/google/gofuzz v1.2.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg=
|
||||
github.com/google/pprof v0.0.0-20250403155104-27863c87afa6 h1:BHT72Gu3keYf3ZEu2J0b1vyeLSOYI8bm5wbJM/8yDe8=
|
||||
github.com/google/pprof v0.0.0-20250403155104-27863c87afa6/go.mod h1:boTsfXsheKC2y+lKOCMpSfarhxDeIzfZG1jqGcPl3cA=
|
||||
github.com/google/s2a-go v0.1.9 h1:LGD7gtMgezd8a/Xak7mEWL0PjoTQFvpRudN895yqKW0=
|
||||
github.com/google/s2a-go v0.1.9/go.mod h1:YA0Ei2ZQL3acow2O62kdp9UlnvMmU7kA6Eutn0dXayM=
|
||||
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
|
||||
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.3.19 h1:mMOE7DN2+p76/EdIrmAy9B9bH+yC4563vmnJ34QR8i4=
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.3.19/go.mod h1:rSEsBUemEBZEexP2y6jPp16LUmUbjmSbcPMQizR0o4k=
|
||||
github.com/googleapis/gax-go/v2 v2.23.0 h1:Tchl7qkvE7Ip3y+ztvNufYFvkfqTe7NfLTYGIdJRLuE=
|
||||
github.com/googleapis/gax-go/v2 v2.23.0/go.mod h1:rBQKOVJCdb8IFEzg+FCwlt1LP/xMDGuqUXhUG+XMXEg=
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.27.7 h1:X+2YciYSxvMQK0UZ7sg45ZVabVZBeBuvMkmuI2V3Fak=
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.27.7/go.mod h1:lW34nIZuQ8UDPdkon5fmfp2l3+ZkQ2me/+oecHYLOII=
|
||||
github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8=
|
||||
@@ -154,22 +170,22 @@ github.com/x448/float16 v0.8.4 h1:qLwI1I70+NjRFUR3zs1JPUCgaCXSh3SW62uAKT1mSBM=
|
||||
github.com/x448/float16 v0.8.4/go.mod h1:14CWIYCyZA/cWjXOioeEpHeN/83MdbZDRQHoFcYsOfg=
|
||||
go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64=
|
||||
go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.65.0 h1:7iP2uCb7sGddAr30RRS6xjKy7AZ2JtTOPA3oolgVSw8=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.65.0/go.mod h1:c7hN3ddxs/z6q9xwvfLPk+UHlWRQyaeR1LdgfL/66l0=
|
||||
go.opentelemetry.io/otel v1.41.0 h1:YlEwVsGAlCvczDILpUXpIpPSL/VPugt7zHThEMLce1c=
|
||||
go.opentelemetry.io/otel v1.41.0/go.mod h1:Yt4UwgEKeT05QbLwbyHXEwhnjxNO6D8L5PQP51/46dE=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.67.0 h1:OyrsyzuttWTSur2qN/Lm0m2a8yqyIjUVBZcxFPuXq2o=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.67.0/go.mod h1:C2NGBr+kAB4bk3xtMXfZ94gqFDtg/GkI7e9zqGh5Beg=
|
||||
go.opentelemetry.io/otel v1.44.0 h1:JjwHmHpA4iZ3wBxluu2fbbE7j4kqlE8jXyAyPXH7HqU=
|
||||
go.opentelemetry.io/otel v1.44.0/go.mod h1:BMgjTHL9WPRlRjL2oZCBTL4whCGtXch2H4BhOPIAyYc=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.40.0 h1:QKdN8ly8zEMrByybbQgv8cWBcdAarwmIPZ6FThrWXJs=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.40.0/go.mod h1:bTdK1nhqF76qiPoCCdyFIV+N/sRHYXYCTQc+3VCi3MI=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.40.0 h1:DvJDOPmSWQHWywQS6lKL+pb8s3gBLOZUtw4N+mavW1I=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.40.0/go.mod h1:EtekO9DEJb4/jRyN4v4Qjc2yA7AtfCBuz2FynRUWTXs=
|
||||
go.opentelemetry.io/otel/metric v1.41.0 h1:rFnDcs4gRzBcsO9tS8LCpgR0dxg4aaxWlJxCno7JlTQ=
|
||||
go.opentelemetry.io/otel/metric v1.41.0/go.mod h1:xPvCwd9pU0VN8tPZYzDZV/BMj9CM9vs00GuBjeKhJps=
|
||||
go.opentelemetry.io/otel/sdk v1.40.0 h1:KHW/jUzgo6wsPh9At46+h4upjtccTmuZCFAc9OJ71f8=
|
||||
go.opentelemetry.io/otel/sdk v1.40.0/go.mod h1:Ph7EFdYvxq72Y8Li9q8KebuYUr2KoeyHx0DRMKrYBUE=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.40.0 h1:mtmdVqgQkeRxHgRv4qhyJduP3fYJRMX4AtAlbuWdCYw=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.40.0/go.mod h1:4Z2bGMf0KSK3uRjlczMOeMhKU2rhUqdWNoKcYrtcBPg=
|
||||
go.opentelemetry.io/otel/trace v1.41.0 h1:Vbk2co6bhj8L59ZJ6/xFTskY+tGAbOnCtQGVVa9TIN0=
|
||||
go.opentelemetry.io/otel/trace v1.41.0/go.mod h1:U1NU4ULCoxeDKc09yCWdWe+3QoyweJcISEVa1RBzOis=
|
||||
go.opentelemetry.io/otel/metric v1.44.0 h1:1w0gILTcHdr3YI+ixLyjemwrVnsMURbTZFrSYCdDdmc=
|
||||
go.opentelemetry.io/otel/metric v1.44.0/go.mod h1:8O7hanEPBNgEMmybD3s2VBKcgWOCsA6tzHBPODAiquo=
|
||||
go.opentelemetry.io/otel/sdk v1.44.0 h1:nHYwb9lK+fJPU/dnT6s7W7Z8itMWyqrnVfbheVYrZ58=
|
||||
go.opentelemetry.io/otel/sdk v1.44.0/go.mod h1:Osuydd3Se74nqjAKxid74N5eC+jfEqfTegHRnq58oK0=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.44.0 h1:3LlKgI+VjbVsjNRFZJZAJ30WjXC5VkNRks6si09iEfI=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.44.0/go.mod h1:5B5pMARnXxKhltooO4xUuCBorl65a4EpnTalObqOigA=
|
||||
go.opentelemetry.io/otel/trace v1.44.0 h1:jxF5CsGYCe74MCRx2X4g7WsY/VBKRqqpNvXlX/6gtIk=
|
||||
go.opentelemetry.io/otel/trace v1.44.0/go.mod h1:oLl1jrMQAVo6v3GAggN+1VH9VIz9iUSvW53sW1Q8PIE=
|
||||
go.opentelemetry.io/proto/otlp v1.9.0 h1:l706jCMITVouPOqEnii2fIAuO3IVGBRPV5ICjceRb/A=
|
||||
go.opentelemetry.io/proto/otlp v1.9.0/go.mod h1:xE+Cx5E/eEHw+ISFkwPLwCZefwVjY+pqKg1qcK03+/4=
|
||||
go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto=
|
||||
@@ -182,36 +198,42 @@ go.yaml.in/yaml/v2 v2.4.3 h1:6gvOSjQoTB3vt1l+CU+tSyi/HOjfOjRLJ4YwYZGwRO0=
|
||||
go.yaml.in/yaml/v2 v2.4.3/go.mod h1:zSxWcmIDjOzPXpjlTTbAsKokqkDNAVtZO0WOMiT90s8=
|
||||
go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc=
|
||||
go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
|
||||
golang.org/x/crypto v0.54.0 h1:YLIA59K4fiNzHzjnZt2tUJQjQtUWfWbeHBqKtk3eScw=
|
||||
golang.org/x/crypto v0.54.0/go.mod h1:KWL8ny2AZdGR2cWmzeHrp2azQPGogOv+HeQaVEXC2dk=
|
||||
golang.org/x/exp v0.0.0-20251219203646-944ab1f22d93 h1:fQsdNF2N+/YewlRZiricy4P1iimyPKZ/xwniHj8Q2a0=
|
||||
golang.org/x/exp v0.0.0-20251219203646-944ab1f22d93/go.mod h1:EPRbTFwzwjXj9NpYyyrvenVh9Y+GFeEvMNh7Xuz7xgU=
|
||||
golang.org/x/mod v0.32.0 h1:9F4d3PHLljb6x//jOyokMv3eX+YDeepZSEo3mFJy93c=
|
||||
golang.org/x/mod v0.32.0/go.mod h1:SgipZ/3h2Ci89DlEtEXWUk/HteuRin+HHhN+WbNhguU=
|
||||
golang.org/x/net v0.49.0 h1:eeHFmOGUTtaaPSGNmjBKpbng9MulQsJURQUAfUwY++o=
|
||||
golang.org/x/net v0.49.0/go.mod h1:/ysNB2EvaqvesRkuLAyjI1ycPZlQHM3q01F02UY/MV8=
|
||||
golang.org/x/oauth2 v0.34.0 h1:hqK/t4AKgbqWkdkcAeI8XLmbK+4m4G5YeQRrmiotGlw=
|
||||
golang.org/x/oauth2 v0.34.0/go.mod h1:lzm5WQJQwKZ3nwavOZ3IS5Aulzxi68dUSgRHujetwEA=
|
||||
golang.org/x/sync v0.19.0 h1:vV+1eWNmZ5geRlYjzm2adRgW2/mcpevXNg50YZtPCE4=
|
||||
golang.org/x/sync v0.19.0/go.mod h1:9KTHXmSnoGruLpwFjVSX0lNNA75CykiMECbovNTZqGI=
|
||||
golang.org/x/sys v0.40.0 h1:DBZZqJ2Rkml6QMQsZywtnjnnGvHza6BTfYFWY9kjEWQ=
|
||||
golang.org/x/sys v0.40.0/go.mod h1:OgkHotnGiDImocRcuBABYBEXf8A9a87e/uXjp9XT3ks=
|
||||
golang.org/x/term v0.39.0 h1:RclSuaJf32jOqZz74CkPA9qFuVTX7vhLlpfj/IGWlqY=
|
||||
golang.org/x/term v0.39.0/go.mod h1:yxzUCTP/U+FzoxfdKmLaA0RV1WgE0VY7hXBwKtY/4ww=
|
||||
golang.org/x/text v0.33.0 h1:B3njUFyqtHDUI5jMn1YIr5B0IE2U0qck04r6d4KPAxE=
|
||||
golang.org/x/text v0.33.0/go.mod h1:LuMebE6+rBincTi9+xWTY8TztLzKHc/9C1uBCG27+q8=
|
||||
golang.org/x/time v0.14.0 h1:MRx4UaLrDotUKUdCIqzPC48t1Y9hANFKIRpNx+Te8PI=
|
||||
golang.org/x/time v0.14.0/go.mod h1:eL/Oa2bBBK0TkX57Fyni+NgnyQQN4LitPmob2Hjnqw4=
|
||||
golang.org/x/tools v0.41.0 h1:a9b8iMweWG+S0OBnlU36rzLp20z1Rp10w+IY2czHTQc=
|
||||
golang.org/x/tools v0.41.0/go.mod h1:XSY6eDqxVNiYgezAVqqCeihT4j1U2CCsqvH3WhQpnlg=
|
||||
golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ=
|
||||
golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0=
|
||||
golang.org/x/net v0.57.0 h1:K5+3DljvIuDG9/Jv9rvyMywYNFCQ9RSUY6OOTTkT+tE=
|
||||
golang.org/x/net v0.57.0/go.mod h1:KpXc8iv+r3XplLAG/f7Jsf9RPszJzdR0f58q9vGOuEU=
|
||||
golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs=
|
||||
golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q=
|
||||
golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
|
||||
golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/term v0.45.0 h1:NwWyBmoJCbfTHpxrWoZ9C6/VxOf7ic219I8xZZFdrf0=
|
||||
golang.org/x/term v0.45.0/go.mod h1:9aqxs0blBcrm/n0L9QW0aRVD+ktan8ssZromtqJC43w=
|
||||
golang.org/x/text v0.40.0 h1:Ub2Z6/xjgF1WrYQz2nuITOEegKFtiIy+rieRJ5lHZKs=
|
||||
golang.org/x/text v0.40.0/go.mod h1:hpnzDAfGV753zIKo+wk3u1bVKCGPbrnF7+7LBF/UHVY=
|
||||
golang.org/x/time v0.15.0 h1:bbrp8t3bGUeFOx08pvsMYRTCVSMk89u4tKbNOZbp88U=
|
||||
golang.org/x/time v0.15.0/go.mod h1:Y4YMaQmXwGQZoFaVFk4YpCt4FLQMYKZe9oeV/f4MSno=
|
||||
golang.org/x/tools v0.47.0 h1:7Kn5x/d1svx/PzryTsqeoZN4TZwqeH5pGWjefhLi/1Q=
|
||||
golang.org/x/tools v0.47.0/go.mod h1:dFHnyTvFWY212G+h7ZY4Vsp/K3U4/7W9TyVaAul8uCA=
|
||||
gomodules.xyz/jsonpatch/v2 v2.4.0 h1:Ci3iUJyx9UeRx7CeFN8ARgGbkESwJK+KB9lLcWxY/Zw=
|
||||
gomodules.xyz/jsonpatch/v2 v2.4.0/go.mod h1:AH3dM2RI6uoBZxn3LVrfvJ3E0/9dG4cSrbuBJT4moAY=
|
||||
gonum.org/v1/gonum v0.16.0 h1:5+ul4Swaf3ESvrOnidPp4GZbzf0mxVQpDCYUQE7OJfk=
|
||||
gonum.org/v1/gonum v0.16.0/go.mod h1:fef3am4MQ93R2HHpKnLk4/Tbh/s0+wqD5nfa6Pnwy4E=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260128011058-8636f8732409 h1:merA0rdPeUV3YIIfHHcH4qBkiQAc1nfCKSI7lB4cV2M=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260128011058-8636f8732409/go.mod h1:fl8J1IvUjCilwZzQowmw2b7HQB2eAuYBabMXzWurF+I=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260128011058-8636f8732409 h1:H86B94AW+VfJWDqFeEbBPhEtHzJwJfTbgE2lZa54ZAQ=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260128011058-8636f8732409/go.mod h1:j9x/tPzZkyxcgEFkiKEEGxfvyumM01BEtsW8xzOahRQ=
|
||||
google.golang.org/grpc v1.79.3 h1:sybAEdRIEtvcD68Gx7dmnwjZKlyfuc61Dyo9pGXXkKE=
|
||||
google.golang.org/grpc v1.79.3/go.mod h1:KmT0Kjez+0dde/v2j9vzwoAScgEPx/Bw1CYChhHLrHQ=
|
||||
gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4=
|
||||
gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E=
|
||||
google.golang.org/api v0.292.0 h1:Ewiwo/GTtiaPZSNAZQUcWLh8AYDEoPmIXyJfeoTSMHU=
|
||||
google.golang.org/api v0.292.0/go.mod h1:07kjmMnFGm2RQuCza2EZM/5N68G/fVvFb1xKjWqoFA0=
|
||||
google.golang.org/genproto v0.0.0-20260319201613-d00831a3d3e7 h1:XzmzkmB14QhVhgnawEVsOn6OFsnpyxNPRY9QV01dNB0=
|
||||
google.golang.org/genproto v0.0.0-20260319201613-d00831a3d3e7/go.mod h1:L43LFes82YgSonw6iTXTxXUX1OlULt4AQtkik4ULL/I=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260630182238-925bb5da69e7 h1:jQ9p21COKWjP3VwuFrNRiiOTMh3mPpN45R7SLrH/HUU=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260630182238-925bb5da69e7/go.mod h1:KqHwBx2upmfa1XSi1WuRvC+2VGCLtooKkfmyvRbUmqA=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260803160001-6ac0973c030d h1:IL4hdHzcUv2l/gcg98/Rj3FbtE6axwqslOW8SW0C+S0=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260803160001-6ac0973c030d/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8=
|
||||
google.golang.org/grpc v1.83.0 h1:JeNZEKJFbQxArAMl+hiytHauacDNqJUllNfmIMmpqnQ=
|
||||
google.golang.org/grpc v1.83.0/go.mod h1:kDyl6SKsiHKt0uylY5gtn5cEjkrIOhQOGDgIc4JGwzQ=
|
||||
google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af h1:+5/Sw3GsDNlEmu7TfklWKPdQ0Ykja5VEmq2i817+jbI=
|
||||
google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||
|
||||
6
hack/providers-dev.yaml
Normal file
6
hack/providers-dev.yaml
Normal file
@@ -0,0 +1,6 @@
|
||||
# Providers config for `make run-dev`: local development against the
|
||||
# current kubeconfig context (e.g. a kind cluster). Only the
|
||||
# kubernetes-pod provider — no cloud credentials needed.
|
||||
providers:
|
||||
- name: kubernetes
|
||||
type: kubernetes
|
||||
175
internal/controller/health_test.go
Normal file
175
internal/controller/health_test.go
Normal file
@@ -0,0 +1,175 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/health"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
type fakeSnapshotter struct {
|
||||
snap health.Snapshot
|
||||
ok bool
|
||||
}
|
||||
|
||||
func (f fakeSnapshotter) Snapshot(types.NamespacedName) (health.Snapshot, bool) {
|
||||
return f.snap, f.ok
|
||||
}
|
||||
|
||||
// TestReconcile_healthRepresentation covers the reconciler's half of the
|
||||
// health split: turning the engine's Snapshot into the Healthy condition,
|
||||
// the latency fields, and ultimately the Ready/Unhealthy phases.
|
||||
func TestReconcile_healthRepresentation(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
probeTime := time.Now()
|
||||
freshHash := specHash(managedProxy(), "")
|
||||
runningStub := func() *stubProvider {
|
||||
return &stubProvider{
|
||||
getInst: &provider.Instance{ID: "stub-id-1", IP: "10.1.2.3", State: provider.StateRunning},
|
||||
}
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
proxy *crawlv1alpha1.Proxy
|
||||
stub *stubProvider
|
||||
health HealthSnapshotter
|
||||
wantPhase crawlv1alpha1.ProxyPhase
|
||||
verify func(t *testing.T, r *ProxyReconciler)
|
||||
}{
|
||||
{
|
||||
name: "running and healthy becomes Ready",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: runningStub(),
|
||||
health: fakeSnapshotter{ok: true, snap: health.Snapshot{
|
||||
Healthy: true, Latency: 37 * time.Millisecond, LastProbe: probeTime,
|
||||
}},
|
||||
wantPhase: crawlv1alpha1.PhaseReady,
|
||||
verify: func(t *testing.T, r *ProxyReconciler) {
|
||||
p := getProxy(t, r)
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionHealthy, metav1.ConditionTrue, ReasonProbeSucceeded)
|
||||
if p.Status.LatencyMillis != 37 {
|
||||
t.Errorf("latencyMillis = %d, want 37", p.Status.LatencyMillis)
|
||||
}
|
||||
if p.Status.LastHealthCheckTime == nil {
|
||||
t.Error("lastHealthCheckTime not set")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "running but unhealthy becomes Unhealthy",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: runningStub(),
|
||||
health: fakeSnapshotter{ok: true, snap: health.Snapshot{
|
||||
Healthy: false, LastProbe: probeTime,
|
||||
LastError: "CONNECT refused", ConsecutiveFailures: 3,
|
||||
}},
|
||||
wantPhase: crawlv1alpha1.PhaseUnhealthy,
|
||||
verify: func(t *testing.T, r *ProxyReconciler) {
|
||||
p := getProxy(t, r)
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionHealthy, metav1.ConditionFalse, ReasonProbeFailed)
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy)
|
||||
if !strings.Contains(cond.Message, "CONNECT refused") {
|
||||
t.Errorf("condition message %q does not carry the probe error", cond.Message)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "no verdict yet stays Provisioning without a Healthy condition",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: runningStub(),
|
||||
health: fakeSnapshotter{ok: false},
|
||||
wantPhase: crawlv1alpha1.PhaseProvisioning,
|
||||
verify: func(t *testing.T, r *ProxyReconciler) {
|
||||
p := getProxy(t, r)
|
||||
if apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy) != nil {
|
||||
t.Error("Healthy condition present without an engine verdict")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "external proxy with a healthy verdict becomes Ready",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Finalizers = nil
|
||||
p.Spec = crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "203.0.113.7"},
|
||||
}
|
||||
}),
|
||||
stub: &stubProvider{},
|
||||
health: fakeSnapshotter{ok: true, snap: health.Snapshot{
|
||||
Healthy: true, Latency: 5 * time.Millisecond, LastProbe: probeTime,
|
||||
}},
|
||||
wantPhase: crawlv1alpha1.PhaseReady,
|
||||
verify: func(t *testing.T, r *ProxyReconciler) {
|
||||
assertCondition(t, getProxy(t, r), crawlv1alpha1.ConditionHealthy, metav1.ConditionTrue, ReasonProbeSucceeded)
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "creating a replacement clears the stale Healthy verdict",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Status.Conditions = []metav1.Condition{{
|
||||
Type: crawlv1alpha1.ConditionHealthy, Status: metav1.ConditionTrue,
|
||||
Reason: ReasonProbeSucceeded, LastTransitionTime: metav1.Now(),
|
||||
}}
|
||||
p.Status.LatencyMillis = 42
|
||||
p.Status.LastHealthCheckTime = &metav1.Time{Time: probeTime}
|
||||
}),
|
||||
stub: &stubProvider{createID: "stub-id-2"},
|
||||
health: fakeSnapshotter{ok: false},
|
||||
wantPhase: crawlv1alpha1.PhaseProvisioning,
|
||||
verify: func(t *testing.T, r *ProxyReconciler) {
|
||||
p := getProxy(t, r)
|
||||
if apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy) != nil {
|
||||
t.Error("stale Healthy condition survived instance creation")
|
||||
}
|
||||
if p.Status.LatencyMillis != 0 || p.Status.LastHealthCheckTime != nil {
|
||||
t.Error("stale latency fields survived instance creation")
|
||||
}
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
r := newTestReconciler(t, tc.stub, tc.proxy)
|
||||
r.Health = tc.health
|
||||
if _, err := doReconcile(t, r); err != nil {
|
||||
t.Fatalf("Reconcile: %v", err)
|
||||
}
|
||||
if got := getProxy(t, r).Status.Phase; got != tc.wantPhase {
|
||||
t.Errorf("phase = %s, want %s", got, tc.wantPhase)
|
||||
}
|
||||
tc.verify(t, r)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A nil Health snapshotter must disable representation entirely.
|
||||
func TestReconcile_nilHealthSnapshotter(t *testing.T) {
|
||||
t.Parallel()
|
||||
freshHash := specHash(managedProxy(), "")
|
||||
r := newTestReconciler(t,
|
||||
&stubProvider{getInst: &provider.Instance{ID: "stub-id-1", IP: "10.1.2.3", State: provider.StateRunning}},
|
||||
managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)))
|
||||
|
||||
if _, err := doReconcile(t, r); err != nil {
|
||||
t.Fatalf("Reconcile: %v", err)
|
||||
}
|
||||
p := getProxy(t, r)
|
||||
if apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy) != nil {
|
||||
t.Error("Healthy condition written with no snapshotter configured")
|
||||
}
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseProvisioning {
|
||||
t.Errorf("phase = %s, want Provisioning", p.Status.Phase)
|
||||
}
|
||||
}
|
||||
@@ -18,46 +18,414 @@ package controller
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/controller"
|
||||
"sigs.k8s.io/controller-runtime/pkg/controller/controllerutil"
|
||||
"sigs.k8s.io/controller-runtime/pkg/event"
|
||||
"sigs.k8s.io/controller-runtime/pkg/handler"
|
||||
logf "sigs.k8s.io/controller-runtime/pkg/log"
|
||||
"sigs.k8s.io/controller-runtime/pkg/reconcile"
|
||||
"sigs.k8s.io/controller-runtime/pkg/source"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/health"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// ProxyReconciler reconciles a Proxy object
|
||||
// HealthSnapshotter provides the current probe verdict for a proxy. The
|
||||
// health engine implements it; the reconciler is its only consumer, turning
|
||||
// snapshots into the Healthy condition — the engine owns health state, the
|
||||
// reconciler owns its representation.
|
||||
type HealthSnapshotter interface {
|
||||
Snapshot(key types.NamespacedName) (health.Snapshot, bool)
|
||||
}
|
||||
|
||||
// ProxyReconciler reconciles Proxy objects as a state machine: every
|
||||
// reconcile derives exactly one action from (spec, status, provider Get),
|
||||
// performs it, and requeues. Status is written at most once per reconcile,
|
||||
// by the deferred patch in Reconcile.
|
||||
type ProxyReconciler struct {
|
||||
client.Client
|
||||
Scheme *runtime.Scheme
|
||||
|
||||
// Providers maps spec.provider values to configured backends.
|
||||
Providers map[string]provider.Provider
|
||||
|
||||
// Health supplies probe verdicts; nil disables health representation
|
||||
// (the Healthy condition simply never appears).
|
||||
Health HealthSnapshotter
|
||||
// HealthEvents, when non-nil, is watched as a raw source so the health
|
||||
// engine can enqueue proxies on status-affecting transitions.
|
||||
HealthEvents <-chan event.GenericEvent
|
||||
|
||||
// Poll intervals are struct fields, never consts, so tests can shrink
|
||||
// them to milliseconds.
|
||||
ProvisioningPoll time.Duration // while waiting for an instance to reach Running
|
||||
DriftPoll time.Duration // between re-checks of a Running instance
|
||||
DeletionPoll time.Duration // while waiting for an instance to disappear
|
||||
QuotaRetry time.Duration // after ErrQuotaExceeded; slow, off the backoff curve
|
||||
RequeueNow time.Duration // "process the next state promptly" (Result.Requeue is deprecated)
|
||||
}
|
||||
|
||||
// +kubebuilder:rbac:groups=crawl.example.com,resources=proxies,verbs=get;list;watch;create;update;patch;delete
|
||||
// +kubebuilder:rbac:groups=crawl.example.com,resources=proxies/status,verbs=get;update;patch
|
||||
// +kubebuilder:rbac:groups=crawl.example.com,resources=proxies/finalizers,verbs=update
|
||||
// +kubebuilder:rbac:groups="",resources=secrets,verbs=get;list;watch
|
||||
// The pod verbs are for the kubernetes-pod provider; cluster-scoped, since
|
||||
// its ListByTag enumerates the operator's Pods across all namespaces.
|
||||
// +kubebuilder:rbac:groups="",resources=pods,verbs=get;list;watch;create;delete
|
||||
|
||||
// Reconcile is part of the main kubernetes reconciliation loop which aims to
|
||||
// move the current state of the cluster closer to the desired state.
|
||||
// TODO(user): Modify the Reconcile function to compare the state specified by
|
||||
// the Proxy object against the actual cluster state, and then
|
||||
// perform operations to make the cluster state reflect the state specified by
|
||||
// the user.
|
||||
//
|
||||
// For more details, check Reconcile and its Result here:
|
||||
// - https://pkg.go.dev/sigs.k8s.io/controller-runtime@v0.24.1/pkg/reconcile
|
||||
func (r *ProxyReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) {
|
||||
_ = logf.FromContext(ctx)
|
||||
// Reconcile fetches the Proxy named by req into p (r.Get fills the struct
|
||||
// through the pointer), dispatches to the delete/external/managed state
|
||||
// machines, and flushes any status change exactly once on the way out.
|
||||
func (r *ProxyReconciler) Reconcile(ctx context.Context, req ctrl.Request) (res ctrl.Result, err error) {
|
||||
var p crawlv1alpha1.Proxy
|
||||
if err := r.Get(ctx, req.NamespacedName, &p); err != nil {
|
||||
return ctrl.Result{}, client.IgnoreNotFound(err)
|
||||
}
|
||||
base := p.DeepCopy()
|
||||
defer func() {
|
||||
// NotFound is expected when this reconcile just removed the last
|
||||
// finalizer and the object is already gone.
|
||||
if perr := r.patchStatusIfChanged(ctx, base, &p); perr != nil && !apierrors.IsNotFound(perr) {
|
||||
err = errors.Join(err, perr)
|
||||
}
|
||||
}()
|
||||
|
||||
// TODO(user): your logic here
|
||||
switch {
|
||||
case !p.DeletionTimestamp.IsZero():
|
||||
return r.reconcileDelete(ctx, &p)
|
||||
case p.Spec.Mode == crawlv1alpha1.ModeExternal:
|
||||
return r.reconcileExternal(ctx, &p)
|
||||
default:
|
||||
return r.reconcileManaged(ctx, &p)
|
||||
}
|
||||
}
|
||||
|
||||
func (r *ProxyReconciler) reconcileManaged(ctx context.Context, p *crawlv1alpha1.Proxy) (ctrl.Result, error) {
|
||||
log := logf.FromContext(ctx)
|
||||
|
||||
if controllerutil.AddFinalizer(p, crawlv1alpha1.FinalizerName) {
|
||||
// The Update event re-triggers reconciliation; provisioning starts
|
||||
// on the next pass, with the finalizer safely persisted first.
|
||||
return ctrl.Result{}, r.Update(ctx, p)
|
||||
}
|
||||
|
||||
// Permanent-failure latch: once this generation has failed permanently,
|
||||
// stop calling the provider until the spec changes.
|
||||
if cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned); cond != nil &&
|
||||
cond.Status == metav1.ConditionFalse && cond.Reason == ReasonPermanentError &&
|
||||
cond.ObservedGeneration == p.Generation {
|
||||
return ctrl.Result{}, nil
|
||||
}
|
||||
|
||||
prov, ok := r.Providers[p.Spec.Provider]
|
||||
if !ok {
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonPermanentError,
|
||||
fmt.Sprintf("provider %q is not configured", p.Spec.Provider))
|
||||
return ctrl.Result{}, nil
|
||||
}
|
||||
|
||||
cloudInit, err := r.resolveCloudInit(ctx, p)
|
||||
if err != nil {
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonCloudInitError, err.Error())
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
hash := specHash(p, cloudInit)
|
||||
|
||||
if p.Status.ProviderID == "" {
|
||||
id, err := prov.Create(ctx, provider.CreateRequest{
|
||||
Name: provider.NameFromUID(p.UID),
|
||||
UID: string(p.UID),
|
||||
Namespace: p.Namespace,
|
||||
ProxyName: p.Name,
|
||||
Placement: placementFrom(p.Spec.Placement),
|
||||
CloudInit: cloudInit,
|
||||
Port: p.EffectivePort(),
|
||||
})
|
||||
if err != nil {
|
||||
return r.providerFailure(p, err)
|
||||
}
|
||||
log.Info("created instance", "provider", p.Spec.Provider, "providerID", id)
|
||||
if err := r.setSpecHash(ctx, p, hash); err != nil {
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
p.Status.ProviderID = id
|
||||
p.Status.IP = ""
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonProvisioning, "instance created; waiting for it to run")
|
||||
// Any Healthy verdict belonged to the previous instance; the health
|
||||
// engine starts fresh for the new one (its state was pruned while
|
||||
// the proxy had no IP), and so must the status.
|
||||
apimeta.RemoveStatusCondition(&p.Status.Conditions, crawlv1alpha1.ConditionHealthy)
|
||||
p.Status.LatencyMillis = 0
|
||||
p.Status.LastHealthCheckTime = nil
|
||||
return ctrl.Result{RequeueAfter: r.ProvisioningPoll}, nil
|
||||
}
|
||||
|
||||
if ann := p.Annotations[crawlv1alpha1.AnnotationSpecHash]; ann != hash {
|
||||
if ann == "" {
|
||||
// Adopt: an instance provisioned before the hash-input struct
|
||||
// gained a field (or by an older operator version) keeps its
|
||||
// instance; replacing the whole fleet on upgrade would be wrong.
|
||||
if err := r.setSpecHash(ctx, p, hash); err != nil {
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
return ctrl.Result{RequeueAfter: r.RequeueNow}, nil
|
||||
}
|
||||
return r.replaceInstance(ctx, p, prov, hash)
|
||||
}
|
||||
|
||||
inst, err := prov.Get(ctx, p.Status.ProviderID)
|
||||
if provider.Class(err) == provider.ErrNotFound {
|
||||
p.Status.ProviderID = ""
|
||||
p.Status.IP = ""
|
||||
return ctrl.Result{RequeueAfter: r.RequeueNow}, nil
|
||||
}
|
||||
if err != nil {
|
||||
return r.providerFailure(p, err)
|
||||
}
|
||||
|
||||
switch inst.State {
|
||||
case provider.StateProvisioning:
|
||||
p.Status.IP = ""
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonProvisioning, "waiting for the instance to run")
|
||||
return ctrl.Result{RequeueAfter: r.ProvisioningPoll}, nil
|
||||
case provider.StateRunning:
|
||||
p.Status.IP = inst.IP
|
||||
setProvisioned(p, metav1.ConditionTrue, ReasonCreated, "instance is running")
|
||||
r.applyHealth(p)
|
||||
return ctrl.Result{RequeueAfter: r.DriftPoll}, nil
|
||||
default: // Stopped, Terminated: cattle, not pets — delete and recreate.
|
||||
if err := prov.Delete(ctx, p.Status.ProviderID); err != nil {
|
||||
return r.providerFailure(p, err)
|
||||
}
|
||||
log.Info("deleting instance for recreation", "providerID", p.Status.ProviderID, "state", inst.State)
|
||||
p.Status.IP = ""
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonRecreating,
|
||||
fmt.Sprintf("instance is %s; deleting it for recreation", inst.State))
|
||||
return ctrl.Result{RequeueAfter: r.DeletionPoll}, nil
|
||||
}
|
||||
}
|
||||
|
||||
// replaceInstance handles a spec-hash mismatch. The replacement instance has
|
||||
// the same deterministic name as the old one (both derive from the CR UID),
|
||||
// so recreating before the old instance is fully gone would hit "already
|
||||
// exists" — hence: delete, poll to NotFound, only then advance the hash and
|
||||
// let the create branch run.
|
||||
func (r *ProxyReconciler) replaceInstance(ctx context.Context, p *crawlv1alpha1.Proxy, prov provider.Provider, hash string) (ctrl.Result, error) {
|
||||
_, err := prov.Get(ctx, p.Status.ProviderID)
|
||||
if provider.Class(err) == provider.ErrNotFound {
|
||||
// Old instance is gone. The Update inside setSpecHash refreshes p
|
||||
// from the server — including status — so the status clear must be
|
||||
// staged after it, or it would be silently overwritten. A crash
|
||||
// between the two writes recovers either way: the create branch's
|
||||
// Create is idempotent by name, and a stale ID resolves to NotFound
|
||||
// again.
|
||||
if err := r.setSpecHash(ctx, p, hash); err != nil {
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
p.Status.ProviderID = ""
|
||||
p.Status.IP = ""
|
||||
return ctrl.Result{RequeueAfter: r.RequeueNow}, nil
|
||||
}
|
||||
if err != nil {
|
||||
return r.providerFailure(p, err)
|
||||
}
|
||||
if err := prov.Delete(ctx, p.Status.ProviderID); err != nil {
|
||||
return r.providerFailure(p, err)
|
||||
}
|
||||
logf.FromContext(ctx).Info("replacing instance after spec change", "providerID", p.Status.ProviderID)
|
||||
p.Status.IP = ""
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonReplacing, "spec changed; deleting the old instance before recreating")
|
||||
return ctrl.Result{RequeueAfter: r.DeletionPoll}, nil
|
||||
}
|
||||
|
||||
func (r *ProxyReconciler) reconcileDelete(ctx context.Context, p *crawlv1alpha1.Proxy) (ctrl.Result, error) {
|
||||
if !controllerutil.ContainsFinalizer(p, crawlv1alpha1.FinalizerName) {
|
||||
return ctrl.Result{}, nil
|
||||
}
|
||||
if p.Status.ProviderID == "" {
|
||||
// Nothing was ever recorded as created; orphan GC reaps any stray
|
||||
// instance a crashed create might have left behind.
|
||||
controllerutil.RemoveFinalizer(p, crawlv1alpha1.FinalizerName)
|
||||
return ctrl.Result{}, r.Update(ctx, p)
|
||||
}
|
||||
prov, ok := r.Providers[p.Spec.Provider]
|
||||
if !ok {
|
||||
return ctrl.Result{}, fmt.Errorf(
|
||||
"provider %q is not configured; cannot clean up instance %s", p.Spec.Provider, p.Status.ProviderID)
|
||||
}
|
||||
_, err := prov.Get(ctx, p.Status.ProviderID)
|
||||
if provider.Class(err) == provider.ErrNotFound {
|
||||
controllerutil.RemoveFinalizer(p, crawlv1alpha1.FinalizerName)
|
||||
return ctrl.Result{}, r.Update(ctx, p)
|
||||
}
|
||||
if err != nil {
|
||||
return r.deletionFailure(err)
|
||||
}
|
||||
if err := prov.Delete(ctx, p.Status.ProviderID); err != nil {
|
||||
return r.deletionFailure(err)
|
||||
}
|
||||
logf.FromContext(ctx).Info("deleting instance", "providerID", p.Status.ProviderID)
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonDeleting, "deleting the instance before removing the finalizer")
|
||||
return ctrl.Result{RequeueAfter: r.DeletionPoll}, nil
|
||||
}
|
||||
|
||||
func (r *ProxyReconciler) reconcileExternal(_ context.Context, p *crawlv1alpha1.Proxy) (ctrl.Result, error) {
|
||||
if p.Spec.Endpoint == nil {
|
||||
// CEL guarantees an endpoint on any object that went through the API
|
||||
// server; tolerate its absence instead of panicking.
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonPermanentError, "external proxy has no endpoint")
|
||||
return ctrl.Result{}, nil
|
||||
}
|
||||
p.Status.IP = p.Spec.Endpoint.Host
|
||||
setProvisioned(p, metav1.ConditionTrue, ReasonExternalEndpoint, "tracking an external endpoint")
|
||||
r.applyHealth(p)
|
||||
return ctrl.Result{}, nil
|
||||
}
|
||||
|
||||
// SetupWithManager sets up the controller with the Manager.
|
||||
// providerFailure translates a classified provider error into the
|
||||
// state-machine's reaction: transient errors ride the workqueue's
|
||||
// exponential backoff, quota errors back off slowly without counting as
|
||||
// errors, and permanent errors latch Failed and stop retrying.
|
||||
func (r *ProxyReconciler) providerFailure(p *crawlv1alpha1.Proxy, err error) (ctrl.Result, error) {
|
||||
switch provider.Class(err) {
|
||||
case provider.ErrQuotaExceeded:
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonQuotaExceeded, err.Error())
|
||||
return ctrl.Result{RequeueAfter: r.QuotaRetry}, nil
|
||||
case provider.ErrPermanent:
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonPermanentError, err.Error())
|
||||
return ctrl.Result{}, nil
|
||||
default:
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
}
|
||||
|
||||
// deletionFailure is providerFailure for the finalizer path, where latching
|
||||
// a permanent failure would wedge the object forever with no retry — keep
|
||||
// retrying instead, visibly, until cleanup succeeds or an operator
|
||||
// intervenes.
|
||||
func (r *ProxyReconciler) deletionFailure(err error) (ctrl.Result, error) {
|
||||
if provider.Class(err) == provider.ErrQuotaExceeded {
|
||||
return ctrl.Result{RequeueAfter: r.QuotaRetry}, nil
|
||||
}
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
|
||||
// resolveCloudInit returns the effective cloud-init user-data, reading the
|
||||
// referenced Secret if one is used. Both the spec hash and CreateRequest see
|
||||
// only resolved content, so rotating a Secret triggers replacement.
|
||||
func (r *ProxyReconciler) resolveCloudInit(ctx context.Context, p *crawlv1alpha1.Proxy) (string, error) {
|
||||
ci := p.Spec.CloudInit
|
||||
if ci == nil {
|
||||
return "", nil
|
||||
}
|
||||
if ci.Inline != "" {
|
||||
return ci.Inline, nil
|
||||
}
|
||||
if ci.SecretRef == nil {
|
||||
return "", nil
|
||||
}
|
||||
key := ci.SecretRef.Key
|
||||
if key == "" {
|
||||
key = crawlv1alpha1.DefaultCloudInitSecretKey
|
||||
}
|
||||
var sec corev1.Secret
|
||||
if err := r.Get(ctx, client.ObjectKey{Namespace: p.Namespace, Name: ci.SecretRef.Name}, &sec); err != nil {
|
||||
return "", fmt.Errorf("resolving cloudInit secret %q: %w", ci.SecretRef.Name, err)
|
||||
}
|
||||
data, ok := sec.Data[key]
|
||||
if !ok {
|
||||
return "", fmt.Errorf("cloudInit secret %q has no key %q", ci.SecretRef.Name, key)
|
||||
}
|
||||
return string(data), nil
|
||||
}
|
||||
|
||||
// setSpecHash persists the spec-hash annotation. Status changes staged on p
|
||||
// are untouched by the Update (they live on the status subresource) and are
|
||||
// flushed by the deferred patch in Reconcile.
|
||||
func (r *ProxyReconciler) setSpecHash(ctx context.Context, p *crawlv1alpha1.Proxy, hash string) error {
|
||||
if p.Annotations[crawlv1alpha1.AnnotationSpecHash] == hash {
|
||||
return nil
|
||||
}
|
||||
if p.Annotations == nil {
|
||||
p.Annotations = map[string]string{}
|
||||
}
|
||||
p.Annotations[crawlv1alpha1.AnnotationSpecHash] = hash
|
||||
return r.Update(ctx, p)
|
||||
}
|
||||
|
||||
func placementFrom(ps *crawlv1alpha1.PlacementSpec) provider.Placement {
|
||||
if ps == nil {
|
||||
return provider.Placement{}
|
||||
}
|
||||
return provider.Placement{
|
||||
Region: ps.Region,
|
||||
Zone: ps.Zone,
|
||||
MachineType: ps.MachineType,
|
||||
Image: ps.Image,
|
||||
}
|
||||
}
|
||||
|
||||
// proxiesForSecret maps a Secret event to the Proxies whose cloudInit
|
||||
// references it, so rotating a Secret re-triggers the replacement check.
|
||||
func (r *ProxyReconciler) proxiesForSecret(ctx context.Context, obj client.Object) []reconcile.Request {
|
||||
var list crawlv1alpha1.ProxyList
|
||||
if err := r.List(ctx, &list, client.InNamespace(obj.GetNamespace())); err != nil {
|
||||
logf.FromContext(ctx).Error(err, "listing proxies for secret event", "secret", obj.GetName())
|
||||
return nil
|
||||
}
|
||||
var reqs []reconcile.Request
|
||||
for i := range list.Items {
|
||||
p := &list.Items[i]
|
||||
if ci := p.Spec.CloudInit; ci != nil && ci.SecretRef != nil && ci.SecretRef.Name == obj.GetName() {
|
||||
reqs = append(reqs, reconcile.Request{NamespacedName: client.ObjectKeyFromObject(p)})
|
||||
}
|
||||
}
|
||||
return reqs
|
||||
}
|
||||
|
||||
// SetupWithManager sets up the controller with the Manager. The Secret watch
|
||||
// only fires for Secrets the manager's cache holds; the composition root
|
||||
// (cmd/main.go) restricts that cache to labelled cloud-init Secrets.
|
||||
func (r *ProxyReconciler) SetupWithManager(mgr ctrl.Manager) error {
|
||||
return ctrl.NewControllerManagedBy(mgr).
|
||||
r.applyDefaults()
|
||||
b := ctrl.NewControllerManagedBy(mgr).
|
||||
For(&crawlv1alpha1.Proxy{}).
|
||||
Named("proxy").
|
||||
Complete(r)
|
||||
Watches(&corev1.Secret{}, handler.EnqueueRequestsFromMapFunc(r.proxiesForSecret)).
|
||||
WithOptions(controller.Options{MaxConcurrentReconciles: 3})
|
||||
if r.HealthEvents != nil {
|
||||
b = b.WatchesRawSource(source.Channel(r.HealthEvents, &handler.EnqueueRequestForObject{}))
|
||||
}
|
||||
return b.Complete(r)
|
||||
}
|
||||
|
||||
func (r *ProxyReconciler) applyDefaults() {
|
||||
if r.ProvisioningPoll == 0 {
|
||||
r.ProvisioningPoll = 10 * time.Second
|
||||
}
|
||||
if r.DriftPoll == 0 {
|
||||
r.DriftPoll = 2 * time.Minute
|
||||
}
|
||||
if r.DeletionPoll == 0 {
|
||||
r.DeletionPoll = 10 * time.Second
|
||||
}
|
||||
if r.QuotaRetry == 0 {
|
||||
r.QuotaRetry = 5 * time.Minute
|
||||
}
|
||||
if r.RequeueNow == 0 {
|
||||
r.RequeueNow = time.Second
|
||||
}
|
||||
}
|
||||
|
||||
@@ -17,71 +17,504 @@ limitations under the License.
|
||||
package controller
|
||||
|
||||
import (
|
||||
"context"
|
||||
"time"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
"k8s.io/apimachinery/pkg/api/errors"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/reconcile"
|
||||
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/health"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
var _ = Describe("Proxy Controller", func() {
|
||||
Context("When reconciling a resource", func() {
|
||||
const (
|
||||
resourceName = "test-resource"
|
||||
resourceNamespace = "default"
|
||||
)
|
||||
// These specs drive the reconciler against a real envtest API server, so
|
||||
// CRD structural defaulting and CEL validation are live — the parts the
|
||||
// fake-client action-table tests can't cover. The provider stays a stub:
|
||||
// envtest has no kubelet or cloud, so instance state is simulated by
|
||||
// mutating the stub between reconciles.
|
||||
var _ = Describe("Proxy controller", func() {
|
||||
const ns = "default"
|
||||
|
||||
ctx := context.Background()
|
||||
|
||||
typeNamespacedName := types.NamespacedName{
|
||||
Name: resourceName,
|
||||
Namespace: resourceNamespace,
|
||||
}
|
||||
proxy := &crawlv1alpha1.Proxy{}
|
||||
|
||||
BeforeEach(func() {
|
||||
By("creating the custom resource for the Kind Proxy")
|
||||
err := k8sClient.Get(ctx, typeNamespacedName, proxy)
|
||||
if err != nil && errors.IsNotFound(err) {
|
||||
resource := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: resourceName,
|
||||
Namespace: resourceNamespace,
|
||||
},
|
||||
// TODO(user): Specify other spec details if needed.
|
||||
}
|
||||
Expect(k8sClient.Create(ctx, resource)).To(Succeed())
|
||||
}
|
||||
})
|
||||
|
||||
AfterEach(func() {
|
||||
// TODO(user): Cleanup logic after each test, like removing the resource instance.
|
||||
resource := &crawlv1alpha1.Proxy{}
|
||||
err := k8sClient.Get(ctx, typeNamespacedName, resource)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
|
||||
By("Cleanup the specific resource instance Proxy")
|
||||
Expect(k8sClient.Delete(ctx, resource)).To(Succeed())
|
||||
})
|
||||
It("should successfully reconcile the resource", func() {
|
||||
By("Reconciling the created resource")
|
||||
controllerReconciler := &ProxyReconciler{
|
||||
newEnvtestReconciler := func(stub *stubProvider) *ProxyReconciler {
|
||||
return &ProxyReconciler{
|
||||
Client: k8sClient,
|
||||
Scheme: k8sClient.Scheme(),
|
||||
Providers: map[string]provider.Provider{"stub": stub},
|
||||
ProvisioningPoll: 50 * time.Millisecond,
|
||||
DriftPoll: 100 * time.Millisecond,
|
||||
DeletionPoll: 50 * time.Millisecond,
|
||||
QuotaRetry: 200 * time.Millisecond,
|
||||
RequeueNow: 10 * time.Millisecond,
|
||||
}
|
||||
}
|
||||
|
||||
_, err := controllerReconciler.Reconcile(ctx, reconcile.Request{
|
||||
NamespacedName: typeNamespacedName,
|
||||
envReconcile := func(r *ProxyReconciler, name string) (ctrl.Result, error) {
|
||||
return r.Reconcile(ctx, ctrl.Request{
|
||||
NamespacedName: types.NamespacedName{Namespace: ns, Name: name},
|
||||
})
|
||||
}
|
||||
|
||||
fetch := func(name string) *crawlv1alpha1.Proxy {
|
||||
p := &crawlv1alpha1.Proxy{}
|
||||
Expect(k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, p)).To(Succeed())
|
||||
return p
|
||||
}
|
||||
|
||||
// cleanup drives a Managed proxy's finalizer to completion so one spec's
|
||||
// leftovers can't leak into another. Registered via DeferCleanup so it
|
||||
// runs even when the spec body fails mid-way.
|
||||
cleanup := func(r *ProxyReconciler, stub *stubProvider, name string) {
|
||||
p := &crawlv1alpha1.Proxy{}
|
||||
err := k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, p)
|
||||
if apierrors.IsNotFound(err) {
|
||||
return
|
||||
}
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
// TODO(user): Add more specific assertions depending on your controller's reconciliation logic.
|
||||
// Example: If you expect a certain status condition after reconciliation, verify it here.
|
||||
Expect(k8sClient.Delete(ctx, p)).To(Succeed())
|
||||
stub.getErr = provider.Wrap(provider.ErrNotFound, "get", "stub", "", nil)
|
||||
for range 3 {
|
||||
if _, err := envReconcile(r, name); err != nil {
|
||||
break
|
||||
}
|
||||
if apierrors.IsNotFound(k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, p)) {
|
||||
return
|
||||
}
|
||||
}
|
||||
Fail("cleanup did not drive the proxy " + name + " to deletion")
|
||||
}
|
||||
|
||||
managedSpec := func() crawlv1alpha1.ProxySpec {
|
||||
return crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged,
|
||||
Provider: "stub",
|
||||
}
|
||||
}
|
||||
|
||||
It("provisions a Managed proxy through to Running", func() {
|
||||
const name = "e2e-provision"
|
||||
stub := &stubProvider{createID: "inst-1"}
|
||||
r := newEnvtestReconciler(stub)
|
||||
DeferCleanup(func() { cleanup(r, stub, name) })
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: managedSpec(),
|
||||
})).To(Succeed())
|
||||
|
||||
By("adding the finalizer on the first pass")
|
||||
res, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res).To(Equal(ctrl.Result{}))
|
||||
Expect(fetch(name).Finalizers).To(ContainElement(crawlv1alpha1.FinalizerName))
|
||||
|
||||
By("creating the instance on the second pass")
|
||||
res, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.ProvisioningPoll))
|
||||
p := fetch(name)
|
||||
Expect(p.Status.ProviderID).To(Equal("inst-1"))
|
||||
Expect(p.Annotations).To(HaveKey(crawlv1alpha1.AnnotationSpecHash))
|
||||
// The real API server defaulted spec.port; the create request must
|
||||
// have seen it.
|
||||
Expect(p.Spec.Port).To(Equal(crawlv1alpha1.DefaultPort))
|
||||
Expect(stub.lastCreate.Port).To(Equal(crawlv1alpha1.DefaultPort))
|
||||
Expect(stub.lastCreate.Name).To(Equal(provider.NameFromUID(p.UID)))
|
||||
|
||||
By("polling while the instance provisions")
|
||||
stub.getInst = &provider.Instance{ID: "inst-1", State: provider.StateProvisioning}
|
||||
res, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.ProvisioningPoll))
|
||||
Expect(fetch(name).Status.Phase).To(Equal(crawlv1alpha1.PhaseProvisioning))
|
||||
|
||||
By("publishing the IP once the instance runs")
|
||||
stub.getInst = &provider.Instance{ID: "inst-1", IP: "10.9.8.7", State: provider.StateRunning}
|
||||
res, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.DriftPoll))
|
||||
p = fetch(name)
|
||||
Expect(p.Status.IP).To(Equal("10.9.8.7"))
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned)
|
||||
Expect(cond).NotTo(BeNil())
|
||||
Expect(cond.Status).To(Equal(metav1.ConditionTrue))
|
||||
Expect(p.Status.ObservedGeneration).To(Equal(p.Generation))
|
||||
})
|
||||
|
||||
It("replaces the instance when the spec changes", func() {
|
||||
const name = "e2e-replace"
|
||||
stub := &stubProvider{createID: "inst-old"}
|
||||
r := newEnvtestReconciler(stub)
|
||||
DeferCleanup(func() { cleanup(r, stub, name) })
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: managedSpec(),
|
||||
})).To(Succeed())
|
||||
_, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
stub.getInst = &provider.Instance{ID: "inst-old", IP: "10.0.0.1", State: provider.StateRunning}
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
oldHash := fetch(name).Annotations[crawlv1alpha1.AnnotationSpecHash]
|
||||
|
||||
By("editing a replacement-triggering field")
|
||||
p := fetch(name)
|
||||
p.Spec.Port = 8080
|
||||
Expect(k8sClient.Update(ctx, p)).To(Succeed())
|
||||
|
||||
By("deleting the old instance first")
|
||||
res, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.DeletionPoll))
|
||||
Expect(stub.deleteCalls).To(Equal(1))
|
||||
p = fetch(name)
|
||||
Expect(p.Annotations[crawlv1alpha1.AnnotationSpecHash]).To(Equal(oldHash),
|
||||
"hash must not advance while the old instance still exists")
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned)
|
||||
Expect(cond.Reason).To(Equal(ReasonReplacing))
|
||||
|
||||
By("advancing the hash once the old instance is gone")
|
||||
stub.getErr = provider.Wrap(provider.ErrNotFound, "get", "stub", "inst-old", nil)
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
p = fetch(name)
|
||||
Expect(p.Status.ProviderID).To(BeEmpty())
|
||||
Expect(p.Annotations[crawlv1alpha1.AnnotationSpecHash]).NotTo(Equal(oldHash))
|
||||
|
||||
By("creating the replacement")
|
||||
stub.createID = "inst-new"
|
||||
stub.getErr = nil
|
||||
stub.getInst = nil
|
||||
res, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.ProvisioningPoll))
|
||||
Expect(stub.createCalls).To(Equal(2))
|
||||
Expect(fetch(name).Status.ProviderID).To(Equal("inst-new"))
|
||||
Expect(stub.lastCreate.Port).To(Equal(int32(8080)))
|
||||
})
|
||||
|
||||
It("cleans up the instance on delete via the finalizer", func() {
|
||||
const name = "e2e-delete"
|
||||
stub := &stubProvider{createID: "inst-del"}
|
||||
r := newEnvtestReconciler(stub)
|
||||
DeferCleanup(func() { cleanup(r, stub, name) })
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: managedSpec(),
|
||||
})).To(Succeed())
|
||||
_, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
stub.getInst = &provider.Instance{ID: "inst-del", IP: "10.0.0.2", State: provider.StateRunning}
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
|
||||
By("deleting the CR — the finalizer holds it")
|
||||
Expect(k8sClient.Delete(ctx, fetch(name))).To(Succeed())
|
||||
res, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.DeletionPoll))
|
||||
Expect(stub.deleteCalls).To(Equal(1))
|
||||
Expect(fetch(name).Status.Phase).To(Equal(crawlv1alpha1.PhaseDeleting))
|
||||
|
||||
By("removing the finalizer once the instance is gone")
|
||||
stub.getErr = provider.Wrap(provider.ErrNotFound, "get", "stub", "inst-del", nil)
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
err = k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &crawlv1alpha1.Proxy{})
|
||||
Expect(apierrors.IsNotFound(err)).To(BeTrue(), "proxy should be fully deleted")
|
||||
})
|
||||
|
||||
It("reaches Ready once the health engine has a verdict", func() {
|
||||
const name = "e2e-ready"
|
||||
stub := &stubProvider{createID: "inst-rdy"}
|
||||
r := newEnvtestReconciler(stub)
|
||||
DeferCleanup(func() { cleanup(r, stub, name) })
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: managedSpec(),
|
||||
})).To(Succeed())
|
||||
_, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
stub.getInst = &provider.Instance{ID: "inst-rdy", IP: "10.3.3.3", State: provider.StateRunning}
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(fetch(name).Status.Phase).To(Equal(crawlv1alpha1.PhaseProvisioning),
|
||||
"no health verdict yet — must not be Ready")
|
||||
|
||||
By("supplying a healthy snapshot")
|
||||
r.Health = fakeSnapshotter{ok: true, snap: health.Snapshot{
|
||||
Healthy: true, Latency: 21 * time.Millisecond, LastProbe: time.Now(),
|
||||
}}
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
p := fetch(name)
|
||||
Expect(p.Status.Phase).To(Equal(crawlv1alpha1.PhaseReady))
|
||||
Expect(p.Status.LatencyMillis).To(Equal(int64(21)))
|
||||
})
|
||||
|
||||
It("treats quota exhaustion as a wait and a permanent error as Failed", func() {
|
||||
const name = "e2e-errors"
|
||||
stub := &stubProvider{
|
||||
createErr: provider.Wrap(provider.ErrQuotaExceeded, "create", "stub", "", nil),
|
||||
}
|
||||
r := newEnvtestReconciler(stub)
|
||||
DeferCleanup(func() { cleanup(r, stub, name) })
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: managedSpec(),
|
||||
})).To(Succeed())
|
||||
_, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
|
||||
By("quota: condition set, slow requeue, phase NOT Failed")
|
||||
res, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred(), "quota must not count as an error (stays off the backoff curve)")
|
||||
Expect(res.RequeueAfter).To(Equal(r.QuotaRetry))
|
||||
p := fetch(name)
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned)
|
||||
Expect(cond.Reason).To(Equal(ReasonQuotaExceeded))
|
||||
Expect(p.Status.Phase).NotTo(Equal(crawlv1alpha1.PhaseFailed))
|
||||
|
||||
By("permanent: phase Failed and no further provider calls")
|
||||
stub.createErr = provider.Wrap(provider.ErrPermanent, "create", "stub", "", nil)
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(fetch(name).Status.Phase).To(Equal(crawlv1alpha1.PhaseFailed))
|
||||
callsAfterLatch := stub.createCalls
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(stub.createCalls).To(Equal(callsAfterLatch), "the latch must stop provider calls")
|
||||
})
|
||||
|
||||
It("adopts an instance when the spec-hash annotation is stripped", func() {
|
||||
const name = "e2e-adopt"
|
||||
stub := &stubProvider{createID: "inst-adopt"}
|
||||
r := newEnvtestReconciler(stub)
|
||||
DeferCleanup(func() { cleanup(r, stub, name) })
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: managedSpec(),
|
||||
})).To(Succeed())
|
||||
_, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
stub.getInst = &provider.Instance{ID: "inst-adopt", IP: "10.4.4.4", State: provider.StateRunning}
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
originalHash := fetch(name).Annotations[crawlv1alpha1.AnnotationSpecHash]
|
||||
Expect(originalHash).NotTo(BeEmpty())
|
||||
|
||||
By("stripping the annotation, as an operator-version upgrade with a changed hash input would")
|
||||
p := fetch(name)
|
||||
delete(p.Annotations, crawlv1alpha1.AnnotationSpecHash)
|
||||
Expect(k8sClient.Update(ctx, p)).To(Succeed())
|
||||
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
p = fetch(name)
|
||||
Expect(p.Annotations[crawlv1alpha1.AnnotationSpecHash]).To(Equal(originalHash), "hash must be restored")
|
||||
Expect(p.Status.ProviderID).To(Equal("inst-adopt"), "adoption must keep the instance")
|
||||
Expect(stub.deleteCalls).To(BeZero(), "adoption must never replace")
|
||||
})
|
||||
|
||||
It("tracks an External proxy without touching providers", func() {
|
||||
const name = "e2e-external"
|
||||
stub := &stubProvider{}
|
||||
r := newEnvtestReconciler(stub)
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "203.0.113.7"},
|
||||
},
|
||||
})).To(Succeed())
|
||||
|
||||
res, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res).To(Equal(ctrl.Result{}))
|
||||
|
||||
p := fetch(name)
|
||||
Expect(p.Status.IP).To(Equal("203.0.113.7"))
|
||||
Expect(p.Finalizers).To(BeEmpty())
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned)
|
||||
Expect(cond).NotTo(BeNil())
|
||||
Expect(cond.Reason).To(Equal(ReasonExternalEndpoint))
|
||||
Expect(stub.createCalls + stub.getCalls + stub.deleteCalls).To(BeZero())
|
||||
|
||||
By("deleting without any finalizer round-trip")
|
||||
Expect(k8sClient.Delete(ctx, p)).To(Succeed())
|
||||
err = k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &crawlv1alpha1.Proxy{})
|
||||
Expect(apierrors.IsNotFound(err)).To(BeTrue())
|
||||
})
|
||||
})
|
||||
|
||||
// These specs assert the CRD's CEL rules and structural defaulting against
|
||||
// the real envtest API server — the fake client runs neither, which is the
|
||||
// documented caveat on the action-table unit tests.
|
||||
var _ = Describe("Proxy CRD validation (CEL)", func() {
|
||||
const ns = "default"
|
||||
|
||||
managed := func(name string) *crawlv1alpha1.Proxy {
|
||||
return &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged,
|
||||
Provider: "stub",
|
||||
},
|
||||
}
|
||||
}
|
||||
external := func(name string) *crawlv1alpha1.Proxy {
|
||||
return &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "203.0.113.9"},
|
||||
},
|
||||
}
|
||||
}
|
||||
mustCreate := func(p *crawlv1alpha1.Proxy) {
|
||||
GinkgoHelper()
|
||||
Expect(k8sClient.Create(ctx, p)).To(Succeed())
|
||||
DeferCleanup(func() { _ = k8sClient.Delete(ctx, p) })
|
||||
}
|
||||
|
||||
It("rejects invalid creates", func() {
|
||||
invalid := []struct {
|
||||
about string
|
||||
spec crawlv1alpha1.ProxySpec
|
||||
want string
|
||||
}{
|
||||
{
|
||||
about: "Managed without provider",
|
||||
spec: crawlv1alpha1.ProxySpec{Mode: crawlv1alpha1.ModeManaged},
|
||||
want: "provider is required when mode is Managed",
|
||||
},
|
||||
{
|
||||
about: "External with provider",
|
||||
spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal, Provider: "stub",
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "203.0.113.9"},
|
||||
},
|
||||
want: "provider must not be set when mode is External",
|
||||
},
|
||||
{
|
||||
about: "External without endpoint",
|
||||
spec: crawlv1alpha1.ProxySpec{Mode: crawlv1alpha1.ModeExternal},
|
||||
want: "endpoint is required when mode is External",
|
||||
},
|
||||
{
|
||||
about: "Managed with endpoint",
|
||||
spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged, Provider: "stub",
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "203.0.113.9"},
|
||||
},
|
||||
want: "endpoint must not be set when mode is Managed",
|
||||
},
|
||||
{
|
||||
about: "cloudInit with both inline and secretRef",
|
||||
spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged, Provider: "stub",
|
||||
CloudInit: &crawlv1alpha1.CloudInitSpec{
|
||||
Inline: "#cloud-config",
|
||||
SecretRef: &crawlv1alpha1.SecretKeySelector{Name: "s"},
|
||||
},
|
||||
},
|
||||
want: "exactly one of inline or secretRef",
|
||||
},
|
||||
{
|
||||
about: "cloudInit with neither inline nor secretRef",
|
||||
spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged, Provider: "stub",
|
||||
CloudInit: &crawlv1alpha1.CloudInitSpec{},
|
||||
},
|
||||
want: "exactly one of inline or secretRef",
|
||||
},
|
||||
}
|
||||
for _, tc := range invalid {
|
||||
p := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "cel-invalid", Namespace: ns},
|
||||
Spec: tc.spec,
|
||||
}
|
||||
err := k8sClient.Create(ctx, p)
|
||||
Expect(err).To(HaveOccurred(), tc.about)
|
||||
Expect(err.Error()).To(ContainSubstring(tc.want), tc.about)
|
||||
}
|
||||
})
|
||||
|
||||
It("rejects mode mutation", func() {
|
||||
p := external("cel-mode-immutable")
|
||||
mustCreate(p)
|
||||
p.Spec = crawlv1alpha1.ProxySpec{Mode: crawlv1alpha1.ModeManaged, Provider: "stub"}
|
||||
err := k8sClient.Update(ctx, p)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err.Error()).To(ContainSubstring("mode is immutable"))
|
||||
})
|
||||
|
||||
It("rejects provider mutation and removal", func() {
|
||||
p := managed("cel-provider-immutable")
|
||||
mustCreate(p)
|
||||
|
||||
p.Spec.Provider = "other"
|
||||
err := k8sClient.Update(ctx, p)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err.Error()).To(ContainSubstring("provider is immutable"))
|
||||
|
||||
// Removal must also be rejected — the has()==has() form exists
|
||||
// exactly because a field-level rule would not fire on absence.
|
||||
// (Dropping provider alone would also trip the required-iff rule,
|
||||
// so flip mode too and check the immutability rules win.)
|
||||
fresh := fetchProxy(ns, "cel-provider-immutable")
|
||||
fresh.Spec.Provider = ""
|
||||
fresh.Spec.Endpoint = &crawlv1alpha1.EndpointSpec{Host: "203.0.113.9"}
|
||||
fresh.Spec.Mode = crawlv1alpha1.ModeExternal
|
||||
err = k8sClient.Update(ctx, fresh)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err.Error()).To(ContainSubstring("immutable"))
|
||||
})
|
||||
|
||||
It("materializes every nested healthCheck default when healthCheck is omitted", func() {
|
||||
p := managed("cel-defaults")
|
||||
mustCreate(p)
|
||||
|
||||
got := fetchProxy(ns, "cel-defaults")
|
||||
// The +kubebuilder:default={} assertion: structural defaulting only
|
||||
// descends into values that exist, so without it a nil healthCheck
|
||||
// would get none of these.
|
||||
hc := got.Spec.HealthCheck
|
||||
Expect(hc).NotTo(BeNil())
|
||||
Expect(hc.ProbeURL).To(Equal(crawlv1alpha1.DefaultProbeURL))
|
||||
Expect(hc.IntervalSeconds).To(Equal(crawlv1alpha1.DefaultHealthCheckIntervalSeconds))
|
||||
Expect(hc.TimeoutSeconds).To(Equal(crawlv1alpha1.DefaultHealthCheckTimeoutSeconds))
|
||||
Expect(hc.FailureThreshold).To(Equal(crawlv1alpha1.DefaultFailureThreshold))
|
||||
Expect(hc.SuccessThreshold).To(Equal(crawlv1alpha1.DefaultSuccessThreshold))
|
||||
Expect(hc.ExpectedStatusCodes).To(Equal(crawlv1alpha1.DefaultExpectedStatusCodes))
|
||||
|
||||
Expect(got.Spec.Port).To(Equal(crawlv1alpha1.DefaultPort))
|
||||
Expect(got.Spec.MaxLeases).NotTo(BeNil())
|
||||
Expect(*got.Spec.MaxLeases).To(Equal(crawlv1alpha1.DefaultMaxLeases))
|
||||
})
|
||||
})
|
||||
|
||||
func fetchProxy(ns, name string) *crawlv1alpha1.Proxy {
|
||||
GinkgoHelper()
|
||||
p := &crawlv1alpha1.Proxy{}
|
||||
Expect(k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, p)).To(Succeed())
|
||||
return p
|
||||
}
|
||||
|
||||
550
internal/controller/reconcile_test.go
Normal file
550
internal/controller/reconcile_test.go
Normal file
@@ -0,0 +1,550 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// Distinct per-interval values so an asserted ctrl.Result is unambiguous
|
||||
// about which action-table row produced it.
|
||||
const (
|
||||
tProvisioningPoll = 11 * time.Second
|
||||
tDriftPoll = 22 * time.Second
|
||||
tDeletionPoll = 33 * time.Second
|
||||
tQuotaRetry = 44 * time.Second
|
||||
tRequeueNow = 55 * time.Millisecond
|
||||
)
|
||||
|
||||
const (
|
||||
testProxyName = "p1"
|
||||
testNamespace = "default"
|
||||
testUID = types.UID("11111111-2222-3333-4444-555555555555")
|
||||
)
|
||||
|
||||
// stubProvider is the plan's in-test Provider stub: a handful of lines, no
|
||||
// config format, no fault-injection surface beyond settable fields.
|
||||
type stubProvider struct {
|
||||
createID string
|
||||
createErr error
|
||||
getInst *provider.Instance
|
||||
getErr error
|
||||
deleteErr error
|
||||
|
||||
createCalls, getCalls, deleteCalls int
|
||||
lastCreate provider.CreateRequest
|
||||
}
|
||||
|
||||
func (s *stubProvider) Create(_ context.Context, req provider.CreateRequest) (string, error) {
|
||||
s.createCalls++
|
||||
s.lastCreate = req
|
||||
if s.createErr != nil {
|
||||
return "", s.createErr
|
||||
}
|
||||
return s.createID, nil
|
||||
}
|
||||
|
||||
func (s *stubProvider) Get(_ context.Context, _ string) (*provider.Instance, error) {
|
||||
s.getCalls++
|
||||
if s.getErr != nil {
|
||||
return nil, s.getErr
|
||||
}
|
||||
return s.getInst, nil
|
||||
}
|
||||
|
||||
func (s *stubProvider) Delete(_ context.Context, _ string) error {
|
||||
s.deleteCalls++
|
||||
return s.deleteErr
|
||||
}
|
||||
|
||||
func (s *stubProvider) ListByTag(context.Context) ([]provider.Instance, error) {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
func notFoundErr() error {
|
||||
return provider.Wrap(provider.ErrNotFound, "get", "stub", "some-id", nil)
|
||||
}
|
||||
|
||||
func testScheme(t *testing.T) *runtime.Scheme {
|
||||
t.Helper()
|
||||
s := runtime.NewScheme()
|
||||
if err := crawlv1alpha1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("adding crawl scheme: %v", err)
|
||||
}
|
||||
if err := corev1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("adding core scheme: %v", err)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// managedProxy returns a Managed proxy that already carries the finalizer —
|
||||
// the state most action-table rows start from. Mutators adjust from there.
|
||||
func managedProxy(mut ...func(*crawlv1alpha1.Proxy)) *crawlv1alpha1.Proxy {
|
||||
p := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: testProxyName,
|
||||
Namespace: testNamespace,
|
||||
UID: testUID,
|
||||
Generation: 1,
|
||||
Finalizers: []string{crawlv1alpha1.FinalizerName},
|
||||
},
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged,
|
||||
Provider: "stub",
|
||||
},
|
||||
}
|
||||
for _, m := range mut {
|
||||
m(p)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
func withProviderID(id string) func(*crawlv1alpha1.Proxy) {
|
||||
return func(p *crawlv1alpha1.Proxy) { p.Status.ProviderID = id }
|
||||
}
|
||||
|
||||
func withSpecHashAnnotation(hash string) func(*crawlv1alpha1.Proxy) {
|
||||
return func(p *crawlv1alpha1.Proxy) {
|
||||
if p.Annotations == nil {
|
||||
p.Annotations = map[string]string{}
|
||||
}
|
||||
p.Annotations[crawlv1alpha1.AnnotationSpecHash] = hash
|
||||
}
|
||||
}
|
||||
|
||||
func deleting() func(*crawlv1alpha1.Proxy) {
|
||||
return func(p *crawlv1alpha1.Proxy) {
|
||||
now := metav1.Now()
|
||||
p.DeletionTimestamp = &now
|
||||
}
|
||||
}
|
||||
|
||||
func newTestReconciler(t *testing.T, stub *stubProvider, objs ...client.Object) *ProxyReconciler {
|
||||
t.Helper()
|
||||
c := fake.NewClientBuilder().
|
||||
WithScheme(testScheme(t)).
|
||||
WithStatusSubresource(&crawlv1alpha1.Proxy{}).
|
||||
WithObjects(objs...).
|
||||
Build()
|
||||
return &ProxyReconciler{
|
||||
Client: c,
|
||||
Providers: map[string]provider.Provider{"stub": stub},
|
||||
ProvisioningPoll: tProvisioningPoll,
|
||||
DriftPoll: tDriftPoll,
|
||||
DeletionPoll: tDeletionPoll,
|
||||
QuotaRetry: tQuotaRetry,
|
||||
RequeueNow: tRequeueNow,
|
||||
}
|
||||
}
|
||||
|
||||
func doReconcile(t *testing.T, r *ProxyReconciler) (ctrl.Result, error) {
|
||||
t.Helper()
|
||||
return r.Reconcile(context.Background(), ctrl.Request{
|
||||
NamespacedName: types.NamespacedName{Namespace: testNamespace, Name: testProxyName},
|
||||
})
|
||||
}
|
||||
|
||||
func getProxy(t *testing.T, r *ProxyReconciler) *crawlv1alpha1.Proxy {
|
||||
t.Helper()
|
||||
var p crawlv1alpha1.Proxy
|
||||
if err := r.Get(context.Background(), types.NamespacedName{Namespace: testNamespace, Name: testProxyName}, &p); err != nil {
|
||||
t.Fatalf("getting proxy: %v", err)
|
||||
}
|
||||
return &p
|
||||
}
|
||||
|
||||
func assertCondition(t *testing.T, p *crawlv1alpha1.Proxy, condType string, status metav1.ConditionStatus, reason string) {
|
||||
t.Helper()
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, condType)
|
||||
if cond == nil {
|
||||
t.Fatalf("condition %s missing, have %+v", condType, p.Status.Conditions)
|
||||
}
|
||||
if cond.Status != status || cond.Reason != reason {
|
||||
t.Errorf("condition %s = %s/%s, want %s/%s", condType, cond.Status, cond.Reason, status, reason)
|
||||
}
|
||||
if cond.ObservedGeneration != p.Generation {
|
||||
t.Errorf("condition %s observedGeneration = %d, want %d", condType, cond.ObservedGeneration, p.Generation)
|
||||
}
|
||||
}
|
||||
|
||||
// TestReconcile_actionTable exercises every row of the plan's action table
|
||||
// by calling Reconcile directly against a fake client. Caveat (documented in
|
||||
// the plan): the fake client runs neither CEL validation nor structural
|
||||
// defaulting — the envtest suite covers those.
|
||||
func TestReconcile_actionTable(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// The fixture's hash: no placement, no cloud-init, defaulted port.
|
||||
freshHash := specHash(managedProxy(), "")
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
proxy *crawlv1alpha1.Proxy
|
||||
extraObjs []client.Object
|
||||
stub *stubProvider
|
||||
wantResult ctrl.Result
|
||||
wantErr bool
|
||||
verify func(t *testing.T, r *ProxyReconciler, stub *stubProvider)
|
||||
}{
|
||||
{
|
||||
name: "managed without finalizer gets one and stops",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Finalizers = nil
|
||||
}),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if len(p.Finalizers) != 1 || p.Finalizers[0] != crawlv1alpha1.FinalizerName {
|
||||
t.Errorf("finalizers = %v, want [%s]", p.Finalizers, crawlv1alpha1.FinalizerName)
|
||||
}
|
||||
if stub.createCalls+stub.getCalls+stub.deleteCalls != 0 {
|
||||
t.Errorf("provider was called before the finalizer was persisted")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "empty providerID creates the instance",
|
||||
proxy: managedProxy(),
|
||||
stub: &stubProvider{createID: "stub-id-1"},
|
||||
wantResult: ctrl.Result{RequeueAfter: tProvisioningPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.ProviderID != "stub-id-1" {
|
||||
t.Errorf("providerID = %q, want stub-id-1", p.Status.ProviderID)
|
||||
}
|
||||
if got := p.Annotations[crawlv1alpha1.AnnotationSpecHash]; got != freshHash {
|
||||
t.Errorf("spec-hash annotation = %q, want %q", got, freshHash)
|
||||
}
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonProvisioning)
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseProvisioning {
|
||||
t.Errorf("phase = %s, want Provisioning", p.Status.Phase)
|
||||
}
|
||||
want := provider.CreateRequest{
|
||||
Name: provider.NameFromUID(testUID),
|
||||
UID: string(testUID),
|
||||
Namespace: testNamespace,
|
||||
ProxyName: testProxyName,
|
||||
Port: crawlv1alpha1.DefaultPort,
|
||||
}
|
||||
if stub.lastCreate != want {
|
||||
t.Errorf("CreateRequest = %+v, want %+v", stub.lastCreate, want)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "provisioning instance polls again",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: &stubProvider{getInst: &provider.Instance{ID: "stub-id-1", State: provider.StateProvisioning}},
|
||||
wantResult: ctrl.Result{RequeueAfter: tProvisioningPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.IP != "" {
|
||||
t.Errorf("ip = %q, want empty while provisioning", p.Status.IP)
|
||||
}
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonProvisioning)
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "running instance publishes IP",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: &stubProvider{
|
||||
getInst: &provider.Instance{ID: "stub-id-1", IP: "10.1.2.3", State: provider.StateRunning},
|
||||
},
|
||||
wantResult: ctrl.Result{RequeueAfter: tDriftPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.IP != "10.1.2.3" {
|
||||
t.Errorf("ip = %q, want 10.1.2.3", p.Status.IP)
|
||||
}
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated)
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseProvisioning {
|
||||
t.Errorf("phase = %s, want Provisioning until a health verdict exists", p.Status.Phase)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "stopped instance is deleted for recreation",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: &stubProvider{getInst: &provider.Instance{ID: "stub-id-1", State: provider.StateStopped}},
|
||||
wantResult: ctrl.Result{RequeueAfter: tDeletionPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
if stub.deleteCalls != 1 {
|
||||
t.Errorf("deleteCalls = %d, want 1", stub.deleteCalls)
|
||||
}
|
||||
assertCondition(t, getProxy(t, r), crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonRecreating)
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "vanished instance clears ID for recreation",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash),
|
||||
func(p *crawlv1alpha1.Proxy) { p.Status.IP = "10.1.2.3" }),
|
||||
stub: &stubProvider{getErr: notFoundErr()},
|
||||
wantResult: ctrl.Result{RequeueAfter: tRequeueNow},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.ProviderID != "" || p.Status.IP != "" {
|
||||
t.Errorf("providerID/ip = %q/%q, want both cleared", p.Status.ProviderID, p.Status.IP)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "hash mismatch deletes the old instance",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation("stale-hash")),
|
||||
stub: &stubProvider{
|
||||
getInst: &provider.Instance{ID: "stub-id-1", IP: "10.1.2.3", State: provider.StateRunning},
|
||||
},
|
||||
wantResult: ctrl.Result{RequeueAfter: tDeletionPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
if stub.deleteCalls != 1 {
|
||||
t.Errorf("deleteCalls = %d, want 1", stub.deleteCalls)
|
||||
}
|
||||
p := getProxy(t, r)
|
||||
// The hash must not advance until the old instance is gone,
|
||||
// or a crash would strand a half-replaced proxy.
|
||||
if got := p.Annotations[crawlv1alpha1.AnnotationSpecHash]; got != "stale-hash" {
|
||||
t.Errorf("spec-hash annotation = %q, want still stale-hash", got)
|
||||
}
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonReplacing)
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "empty annotation adopts instead of replacing",
|
||||
proxy: managedProxy(withProviderID("stub-id-1")),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{RequeueAfter: tRequeueNow},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if got := p.Annotations[crawlv1alpha1.AnnotationSpecHash]; got != freshHash {
|
||||
t.Errorf("spec-hash annotation = %q, want %q", got, freshHash)
|
||||
}
|
||||
if p.Status.ProviderID != "stub-id-1" {
|
||||
t.Errorf("providerID = %q, want untouched stub-id-1", p.Status.ProviderID)
|
||||
}
|
||||
if stub.deleteCalls != 0 {
|
||||
t.Errorf("deleteCalls = %d, want 0 — adoption must not replace", stub.deleteCalls)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "mismatch with instance gone advances the hash",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation("stale-hash")),
|
||||
stub: &stubProvider{getErr: notFoundErr()},
|
||||
wantResult: ctrl.Result{RequeueAfter: tRequeueNow},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.ProviderID != "" {
|
||||
t.Errorf("providerID = %q, want cleared", p.Status.ProviderID)
|
||||
}
|
||||
if got := p.Annotations[crawlv1alpha1.AnnotationSpecHash]; got != freshHash {
|
||||
t.Errorf("spec-hash annotation = %q, want advanced to %q", got, freshHash)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "deletion with no providerID removes the finalizer",
|
||||
proxy: managedProxy(deleting()),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
assertProxyGone(t, r)
|
||||
if stub.deleteCalls != 0 {
|
||||
t.Errorf("deleteCalls = %d, want 0", stub.deleteCalls)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "deletion with instance already gone removes the finalizer",
|
||||
proxy: managedProxy(deleting(), withProviderID("stub-id-1")),
|
||||
stub: &stubProvider{getErr: notFoundErr()},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
assertProxyGone(t, r)
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "deletion deletes the instance and polls",
|
||||
proxy: managedProxy(deleting(), withProviderID("stub-id-1")),
|
||||
stub: &stubProvider{
|
||||
getInst: &provider.Instance{ID: "stub-id-1", State: provider.StateRunning},
|
||||
},
|
||||
wantResult: ctrl.Result{RequeueAfter: tDeletionPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
if stub.deleteCalls != 1 {
|
||||
t.Errorf("deleteCalls = %d, want 1", stub.deleteCalls)
|
||||
}
|
||||
p := getProxy(t, r)
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseDeleting {
|
||||
t.Errorf("phase = %s, want Deleting", p.Status.Phase)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "external proxy tracks its endpoint without a finalizer",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Finalizers = nil
|
||||
p.Spec = crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "203.0.113.7", Port: 8080},
|
||||
}
|
||||
}),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.IP != "203.0.113.7" {
|
||||
t.Errorf("ip = %q, want the endpoint host", p.Status.IP)
|
||||
}
|
||||
if len(p.Finalizers) != 0 {
|
||||
t.Errorf("finalizers = %v, want none on External", p.Finalizers)
|
||||
}
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonExternalEndpoint)
|
||||
if stub.createCalls+stub.getCalls+stub.deleteCalls != 0 {
|
||||
t.Errorf("provider was called for an External proxy")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "quota error backs off slowly without failing",
|
||||
proxy: managedProxy(),
|
||||
stub: &stubProvider{
|
||||
createErr: provider.Wrap(provider.ErrQuotaExceeded, "create", "stub", "", nil),
|
||||
},
|
||||
wantResult: ctrl.Result{RequeueAfter: tQuotaRetry},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonQuotaExceeded)
|
||||
if p.Status.Phase == crawlv1alpha1.PhaseFailed {
|
||||
t.Errorf("phase = Failed, want anything but — quota is a wait, not a failure")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "permanent error latches Failed and stops calling the provider",
|
||||
proxy: managedProxy(),
|
||||
stub: &stubProvider{
|
||||
createErr: provider.Wrap(provider.ErrPermanent, "create", "stub", "", nil),
|
||||
},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonPermanentError)
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseFailed {
|
||||
t.Errorf("phase = %s, want Failed", p.Status.Phase)
|
||||
}
|
||||
if res, err := doReconcile(t, r); err != nil || res != (ctrl.Result{}) {
|
||||
t.Errorf("second reconcile = %+v, %v; want empty result, nil", res, err)
|
||||
}
|
||||
if stub.createCalls != 1 {
|
||||
t.Errorf("createCalls = %d after latch, want 1", stub.createCalls)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "transient error is returned for workqueue backoff",
|
||||
proxy: managedProxy(),
|
||||
stub: &stubProvider{
|
||||
createErr: provider.Wrap(provider.ErrTransient, "create", "stub", "", nil),
|
||||
},
|
||||
wantResult: ctrl.Result{},
|
||||
wantErr: true,
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.Phase != crawlv1alpha1.PhasePending {
|
||||
t.Errorf("phase = %s, want still Pending", p.Status.Phase)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "unconfigured provider is a permanent failure",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Spec.Provider = "no-such-provider"
|
||||
}),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonPermanentError)
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseFailed {
|
||||
t.Errorf("phase = %s, want Failed", p.Status.Phase)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "cloud-init secret is resolved into the create request",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Spec.CloudInit = &crawlv1alpha1.CloudInitSpec{
|
||||
SecretRef: &crawlv1alpha1.SecretKeySelector{Name: "ci-secret"},
|
||||
}
|
||||
}),
|
||||
extraObjs: []client.Object{
|
||||
&corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "ci-secret", Namespace: testNamespace},
|
||||
Data: map[string][]byte{"user-data": []byte("#cloud-config\npackages: [squid]")},
|
||||
},
|
||||
},
|
||||
stub: &stubProvider{createID: "stub-id-1"},
|
||||
wantResult: ctrl.Result{RequeueAfter: tProvisioningPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
if want := "#cloud-config\npackages: [squid]"; stub.lastCreate.CloudInit != want {
|
||||
t.Errorf("CreateRequest.CloudInit = %q, want the resolved secret content", stub.lastCreate.CloudInit)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "missing cloud-init secret errors and marks the condition",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Spec.CloudInit = &crawlv1alpha1.CloudInitSpec{
|
||||
SecretRef: &crawlv1alpha1.SecretKeySelector{Name: "absent"},
|
||||
}
|
||||
}),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{},
|
||||
wantErr: true,
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
assertCondition(t, getProxy(t, r), crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonCloudInitError)
|
||||
if stub.createCalls != 0 {
|
||||
t.Errorf("createCalls = %d, want 0 with unresolved cloud-init", stub.createCalls)
|
||||
}
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
objs := append([]client.Object{tc.proxy}, tc.extraObjs...)
|
||||
r := newTestReconciler(t, tc.stub, objs...)
|
||||
res, err := doReconcile(t, r)
|
||||
if (err != nil) != tc.wantErr {
|
||||
t.Fatalf("Reconcile error = %v, wantErr %v", err, tc.wantErr)
|
||||
}
|
||||
if res != tc.wantResult {
|
||||
t.Errorf("Result = %+v, want %+v", res, tc.wantResult)
|
||||
}
|
||||
tc.verify(t, r, tc.stub)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func assertProxyGone(t *testing.T, r *ProxyReconciler) {
|
||||
t.Helper()
|
||||
var p crawlv1alpha1.Proxy
|
||||
err := r.Get(context.Background(), types.NamespacedName{Namespace: testNamespace, Name: testProxyName}, &p)
|
||||
if !apierrors.IsNotFound(err) {
|
||||
t.Errorf("proxy still exists (err=%v), want NotFound after finalizer removal", err)
|
||||
}
|
||||
}
|
||||
41
internal/controller/spechash.go
Normal file
41
internal/controller/spechash.go
Normal file
@@ -0,0 +1,41 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
// specHashInput is the explicit set of fields whose change requires
|
||||
// replacing the instance. Deliberately not ProxySpec wholesale: adding a
|
||||
// spec field that doesn't affect the VM (attributes, maxLeases, healthCheck)
|
||||
// must not churn the fleet on operator upgrade.
|
||||
type specHashInput struct {
|
||||
Placement *crawlv1alpha1.PlacementSpec `json:"placement,omitempty"`
|
||||
CloudInit string `json:"cloudInit,omitempty"`
|
||||
Port int32 `json:"port"`
|
||||
}
|
||||
|
||||
// specHash returns the hex SHA-256 of the replacement-triggering spec
|
||||
// fields. cloudInit is the already-resolved content, so rotating a
|
||||
// referenced Secret changes the hash even though the spec is untouched.
|
||||
func specHash(p *crawlv1alpha1.Proxy, cloudInit string) string {
|
||||
in := specHashInput{
|
||||
Placement: p.Spec.Placement,
|
||||
CloudInit: cloudInit,
|
||||
Port: p.EffectivePort(),
|
||||
}
|
||||
// nil and empty placement mean the same thing; hash them identically.
|
||||
if in.Placement != nil && *in.Placement == (crawlv1alpha1.PlacementSpec{}) {
|
||||
in.Placement = nil
|
||||
}
|
||||
b, err := json.Marshal(in)
|
||||
if err != nil {
|
||||
// A struct of strings and an int32 cannot fail to marshal.
|
||||
panic(err)
|
||||
}
|
||||
sum := sha256.Sum256(b)
|
||||
return hex.EncodeToString(sum[:])
|
||||
}
|
||||
118
internal/controller/spechash_test.go
Normal file
118
internal/controller/spechash_test.go
Normal file
@@ -0,0 +1,118 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"testing"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
func hashProxy(mut ...func(*crawlv1alpha1.Proxy)) *crawlv1alpha1.Proxy {
|
||||
p := &crawlv1alpha1.Proxy{
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged,
|
||||
Provider: "stub",
|
||||
Port: 3128,
|
||||
Placement: &crawlv1alpha1.PlacementSpec{
|
||||
Zone: "europe-west1-b",
|
||||
MachineType: "e2-micro",
|
||||
},
|
||||
},
|
||||
}
|
||||
for _, m := range mut {
|
||||
m(p)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
func TestSpecHash_stability(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
p := hashProxy()
|
||||
h1 := specHash(p, "cloud-init-content")
|
||||
h2 := specHash(p.DeepCopy(), "cloud-init-content")
|
||||
if h1 != h2 {
|
||||
t.Errorf("same input hashed differently: %s vs %s", h1, h2)
|
||||
}
|
||||
if !regexp.MustCompile(`^[0-9a-f]{64}$`).MatchString(h1) {
|
||||
t.Errorf("hash %q is not hex SHA-256", h1)
|
||||
}
|
||||
|
||||
// Fields outside the replacement set must not affect the hash — that is
|
||||
// the whole point of an explicit hash-input struct.
|
||||
q := hashProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Spec.Attributes = map[string]string{"geo": "eu"}
|
||||
five := int32(5)
|
||||
p.Spec.MaxLeases = &five
|
||||
p.Spec.HealthCheck = &crawlv1alpha1.HealthCheckSpec{IntervalSeconds: 60}
|
||||
})
|
||||
if specHash(q, "cloud-init-content") != h1 {
|
||||
t.Error("non-replacement spec fields changed the hash")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSpecHash_normalization(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
nilPlacement := hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Placement = nil })
|
||||
emptyPlacement := hashProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Spec.Placement = &crawlv1alpha1.PlacementSpec{}
|
||||
})
|
||||
if specHash(nilPlacement, "") != specHash(emptyPlacement, "") {
|
||||
t.Error("nil and empty placement hashed differently")
|
||||
}
|
||||
|
||||
// An unset port and an explicit default port mean the same instance.
|
||||
unsetPort := hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Port = 0 })
|
||||
defaultPort := hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Port = crawlv1alpha1.DefaultPort })
|
||||
if specHash(unsetPort, "") != specHash(defaultPort, "") {
|
||||
t.Error("unset port and explicit default port hashed differently")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSpecHash_sensitivity(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
base := specHash(hashProxy(), "cloud-init")
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
proxy *crawlv1alpha1.Proxy
|
||||
cloudInit string
|
||||
}{
|
||||
{
|
||||
name: "port change",
|
||||
proxy: hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Port = 8080 }),
|
||||
cloudInit: "cloud-init",
|
||||
},
|
||||
{
|
||||
name: "zone change",
|
||||
proxy: hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Placement.Zone = "us-east1-c" }),
|
||||
cloudInit: "cloud-init",
|
||||
},
|
||||
{
|
||||
name: "machine type change",
|
||||
proxy: hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Placement.MachineType = "e2-small" }),
|
||||
cloudInit: "cloud-init",
|
||||
},
|
||||
{
|
||||
name: "image change",
|
||||
proxy: hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Placement.Image = "debian-13" }),
|
||||
cloudInit: "cloud-init",
|
||||
},
|
||||
{
|
||||
name: "resolved cloud-init change (secret rotation)",
|
||||
proxy: hashProxy(),
|
||||
cloudInit: "rotated-cloud-init",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
if specHash(tc.proxy, tc.cloudInit) == base {
|
||||
t.Error("hash did not change")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
121
internal/controller/status.go
Normal file
121
internal/controller/status.go
Normal file
@@ -0,0 +1,121 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"k8s.io/apimachinery/pkg/api/equality"
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
// Reasons used on the Provisioned condition, plus the two the reconciler
|
||||
// writes on the Healthy condition when representing the health engine's
|
||||
// verdict (the engine owns the state; only the reconciler writes status).
|
||||
const (
|
||||
ReasonProvisioning = "Provisioning"
|
||||
ReasonCreated = "Created"
|
||||
ReasonReplacing = "Replacing"
|
||||
ReasonRecreating = "Recreating"
|
||||
ReasonQuotaExceeded = "QuotaExceeded"
|
||||
ReasonPermanentError = "PermanentError"
|
||||
ReasonCloudInitError = "CloudInitError"
|
||||
ReasonExternalEndpoint = "ExternalEndpoint"
|
||||
ReasonDeleting = "Deleting"
|
||||
|
||||
ReasonProbeSucceeded = "ProbeSucceeded"
|
||||
ReasonProbeFailed = "ProbeFailed"
|
||||
)
|
||||
|
||||
// setProvisioned stages the Provisioned condition on p. Nothing is written
|
||||
// to the API server here; the deferred patch in Reconcile flushes it.
|
||||
// ObservedGeneration is passed explicitly — SetStatusCondition does not
|
||||
// populate it, and without it every condition would report generation 0.
|
||||
func setProvisioned(p *crawlv1alpha1.Proxy, status metav1.ConditionStatus, reason, message string) {
|
||||
apimeta.SetStatusCondition(&p.Status.Conditions, metav1.Condition{
|
||||
Type: crawlv1alpha1.ConditionProvisioned,
|
||||
Status: status,
|
||||
Reason: reason,
|
||||
Message: message,
|
||||
ObservedGeneration: p.Generation,
|
||||
})
|
||||
}
|
||||
|
||||
// applyHealth stages the Healthy condition and the latency fields from the
|
||||
// health engine's current snapshot. Called only from states where the proxy
|
||||
// is reachable (Running, External); everywhere else the condition is either
|
||||
// left as-is or removed by the create branch.
|
||||
func (r *ProxyReconciler) applyHealth(p *crawlv1alpha1.Proxy) {
|
||||
if r.Health == nil {
|
||||
return
|
||||
}
|
||||
snap, ok := r.Health.Snapshot(client.ObjectKeyFromObject(p))
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
cond := metav1.Condition{
|
||||
Type: crawlv1alpha1.ConditionHealthy,
|
||||
ObservedGeneration: p.Generation,
|
||||
}
|
||||
if snap.Healthy {
|
||||
cond.Status = metav1.ConditionTrue
|
||||
cond.Reason = ReasonProbeSucceeded
|
||||
cond.Message = "probe succeeded through the proxy"
|
||||
} else {
|
||||
cond.Status = metav1.ConditionFalse
|
||||
cond.Reason = ReasonProbeFailed
|
||||
cond.Message = fmt.Sprintf("%d consecutive probe failures; last: %s",
|
||||
snap.ConsecutiveFailures, snap.LastError)
|
||||
}
|
||||
apimeta.SetStatusCondition(&p.Status.Conditions, cond)
|
||||
p.Status.LatencyMillis = snap.Latency.Milliseconds()
|
||||
if !snap.LastProbe.IsZero() {
|
||||
p.Status.LastHealthCheckTime = &metav1.Time{Time: snap.LastProbe}
|
||||
}
|
||||
}
|
||||
|
||||
// computePhase derives status.phase from deletionTimestamp and the
|
||||
// Provisioned/Healthy conditions. Pure, so the truth table is unit-testable.
|
||||
func computePhase(p *crawlv1alpha1.Proxy) crawlv1alpha1.ProxyPhase {
|
||||
if !p.DeletionTimestamp.IsZero() {
|
||||
return crawlv1alpha1.PhaseDeleting
|
||||
}
|
||||
prov := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned)
|
||||
if prov == nil {
|
||||
return crawlv1alpha1.PhasePending
|
||||
}
|
||||
if prov.Status != metav1.ConditionTrue {
|
||||
// Quota exhaustion is a slow-retry wait, not a terminal state — only
|
||||
// a permanent error latches Failed.
|
||||
if prov.Reason == ReasonPermanentError {
|
||||
return crawlv1alpha1.PhaseFailed
|
||||
}
|
||||
return crawlv1alpha1.PhaseProvisioning
|
||||
}
|
||||
healthy := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy)
|
||||
switch {
|
||||
case healthy == nil || healthy.Status == metav1.ConditionUnknown:
|
||||
// Provisioned but no health verdict yet: still being brought into
|
||||
// service.
|
||||
return crawlv1alpha1.PhaseProvisioning
|
||||
case healthy.Status == metav1.ConditionTrue:
|
||||
return crawlv1alpha1.PhaseReady
|
||||
default:
|
||||
return crawlv1alpha1.PhaseUnhealthy
|
||||
}
|
||||
}
|
||||
|
||||
// patchStatusIfChanged recomputes the derived status fields and issues one
|
||||
// status patch — or none, when nothing changed. This is the only place the
|
||||
// reconciler writes status.
|
||||
func (r *ProxyReconciler) patchStatusIfChanged(ctx context.Context, base, p *crawlv1alpha1.Proxy) error {
|
||||
p.Status.ObservedGeneration = p.Generation
|
||||
p.Status.Phase = computePhase(p)
|
||||
if equality.Semantic.DeepEqual(base.Status, p.Status) {
|
||||
return nil
|
||||
}
|
||||
return r.Status().Patch(ctx, p, client.MergeFrom(base))
|
||||
}
|
||||
112
internal/controller/status_test.go
Normal file
112
internal/controller/status_test.go
Normal file
@@ -0,0 +1,112 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
func TestComputePhase_truthTable(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cond := func(condType string, status metav1.ConditionStatus, reason string) metav1.Condition {
|
||||
return metav1.Condition{Type: condType, Status: status, Reason: reason}
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
deleting bool
|
||||
conditions []metav1.Condition
|
||||
want crawlv1alpha1.ProxyPhase
|
||||
}{
|
||||
{
|
||||
name: "deletionTimestamp wins over everything",
|
||||
deleting: true,
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated),
|
||||
cond(crawlv1alpha1.ConditionHealthy, metav1.ConditionTrue, "Probing"),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseDeleting,
|
||||
},
|
||||
{
|
||||
name: "no conditions is Pending",
|
||||
want: crawlv1alpha1.PhasePending,
|
||||
},
|
||||
{
|
||||
name: "provisioning in progress",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonProvisioning),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseProvisioning,
|
||||
},
|
||||
{
|
||||
name: "replacing counts as provisioning",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonReplacing),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseProvisioning,
|
||||
},
|
||||
{
|
||||
name: "quota exhaustion is not Failed",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonQuotaExceeded),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseProvisioning,
|
||||
},
|
||||
{
|
||||
name: "permanent error is Failed",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonPermanentError),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseFailed,
|
||||
},
|
||||
{
|
||||
name: "provisioned without a health verdict stays Provisioning",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseProvisioning,
|
||||
},
|
||||
{
|
||||
name: "provisioned with Healthy Unknown stays Provisioning",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated),
|
||||
cond(crawlv1alpha1.ConditionHealthy, metav1.ConditionUnknown, "NoProbeYet"),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseProvisioning,
|
||||
},
|
||||
{
|
||||
name: "provisioned and healthy is Ready",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated),
|
||||
cond(crawlv1alpha1.ConditionHealthy, metav1.ConditionTrue, "Probing"),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseReady,
|
||||
},
|
||||
{
|
||||
name: "provisioned but unhealthy is Unhealthy",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated),
|
||||
cond(crawlv1alpha1.ConditionHealthy, metav1.ConditionFalse, "ProbeFailed"),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseUnhealthy,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := &crawlv1alpha1.Proxy{}
|
||||
p.Status.Conditions = tc.conditions
|
||||
if tc.deleting {
|
||||
now := metav1.Now()
|
||||
p.DeletionTimestamp = &now
|
||||
}
|
||||
if got := computePhase(p); got != tc.want {
|
||||
t.Errorf("computePhase() = %s, want %s", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -49,6 +49,9 @@ var (
|
||||
)
|
||||
|
||||
func TestControllers(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping envtest suite in -short mode")
|
||||
}
|
||||
RegisterFailHandler(Fail)
|
||||
|
||||
RunSpecs(t, "Controller Suite")
|
||||
|
||||
241
internal/discovery/handlers.go
Normal file
241
internal/discovery/handlers.go
Normal file
@@ -0,0 +1,241 @@
|
||||
package discovery
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"slices"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/lease"
|
||||
)
|
||||
|
||||
// proxyView is the wire shape of one proxy in list and lease responses.
|
||||
type proxyView struct {
|
||||
ID string `json:"id"` // namespace/name
|
||||
IP string `json:"ip"`
|
||||
Port int32 `json:"port"`
|
||||
Attributes map[string]string `json:"attributes,omitempty"`
|
||||
Phase string `json:"phase"`
|
||||
Healthy bool `json:"healthy"`
|
||||
LatencyMillis int64 `json:"latencyMillis"`
|
||||
ActiveLeases int `json:"activeLeases"`
|
||||
MaxLeases int32 `json:"maxLeases"`
|
||||
}
|
||||
|
||||
func viewOf(p *crawlv1alpha1.Proxy, activeLeases int) proxyView {
|
||||
return proxyView{
|
||||
ID: client.ObjectKeyFromObject(p).String(),
|
||||
IP: p.EffectiveHost(),
|
||||
Port: p.EffectivePort(),
|
||||
Attributes: p.Spec.Attributes,
|
||||
Phase: string(p.Status.Phase),
|
||||
Healthy: isHealthy(p),
|
||||
LatencyMillis: p.Status.LatencyMillis,
|
||||
ActiveLeases: activeLeases,
|
||||
MaxLeases: p.MaxLeasesOrDefault(),
|
||||
}
|
||||
}
|
||||
|
||||
func isHealthy(p *crawlv1alpha1.Proxy) bool {
|
||||
return apimeta.IsStatusConditionTrue(p.Status.Conditions, crawlv1alpha1.ConditionHealthy)
|
||||
}
|
||||
|
||||
// matchesAttributes is spec.attributes equality: every selector pair must
|
||||
// be present verbatim.
|
||||
func matchesAttributes(p *crawlv1alpha1.Proxy, selector map[string]string) bool {
|
||||
for k, v := range selector {
|
||||
if p.Spec.Attributes[k] != v {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// GET /v1/proxies?attr.<key>=<value>&healthy=true|false
|
||||
func (s *Server) handleListProxies(w http.ResponseWriter, r *http.Request) {
|
||||
selector := map[string]string{}
|
||||
var healthyFilter *bool
|
||||
for key, values := range r.URL.Query() {
|
||||
switch {
|
||||
case key == "healthy":
|
||||
switch values[0] {
|
||||
case "true":
|
||||
healthyFilter = ptr(true)
|
||||
case "false":
|
||||
healthyFilter = ptr(false)
|
||||
default:
|
||||
writeError(w, http.StatusBadRequest, "invalid_query",
|
||||
fmt.Sprintf("healthy must be true or false, got %q", values[0]))
|
||||
return
|
||||
}
|
||||
case strings.HasPrefix(key, "attr."):
|
||||
selector[strings.TrimPrefix(key, "attr.")] = values[0]
|
||||
}
|
||||
}
|
||||
|
||||
var list crawlv1alpha1.ProxyList
|
||||
if err := s.Reader.List(r.Context(), &list); err != nil {
|
||||
s.log.Error(err, "listing proxies")
|
||||
writeError(w, http.StatusInternalServerError, "internal", "listing proxies failed")
|
||||
return
|
||||
}
|
||||
counts := s.Store.Counts()
|
||||
|
||||
views := []proxyView{}
|
||||
for i := range list.Items {
|
||||
p := &list.Items[i]
|
||||
if !p.DeletionTimestamp.IsZero() || !matchesAttributes(p, selector) {
|
||||
continue
|
||||
}
|
||||
v := viewOf(p, counts[client.ObjectKeyFromObject(p).String()])
|
||||
if healthyFilter != nil && v.Healthy != *healthyFilter {
|
||||
continue
|
||||
}
|
||||
views = append(views, v)
|
||||
}
|
||||
slices.SortFunc(views, func(a, b proxyView) int { return strings.Compare(a.ID, b.ID) })
|
||||
|
||||
writeJSON(w, http.StatusOK, map[string]any{"proxies": views, "count": len(views)})
|
||||
}
|
||||
|
||||
type leaseRequest struct {
|
||||
Selector map[string]string `json:"selector"`
|
||||
TTLSeconds int64 `json:"ttlSeconds"`
|
||||
Target string `json:"target"`
|
||||
}
|
||||
|
||||
type leaseResponse struct {
|
||||
LeaseID string `json:"leaseID"`
|
||||
Proxy proxyView `json:"proxy"`
|
||||
ExpiresAt time.Time `json:"expiresAt"`
|
||||
TTLSeconds int64 `json:"ttlSeconds"`
|
||||
}
|
||||
|
||||
// POST /v1/leases
|
||||
func (s *Server) handleAcquireLease(w http.ResponseWriter, r *http.Request) {
|
||||
var req leaseRequest
|
||||
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid_body", err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
ttl := time.Duration(req.TTLSeconds) * time.Second
|
||||
if req.TTLSeconds == 0 {
|
||||
ttl = defaultTTL
|
||||
}
|
||||
if ttl < 0 || ttl > s.MaxLeaseTTL {
|
||||
writeError(w, http.StatusBadRequest, "invalid_ttl",
|
||||
fmt.Sprintf("ttlSeconds must be between 1 and %d", int64(s.MaxLeaseTTL.Seconds())))
|
||||
return
|
||||
}
|
||||
|
||||
var list crawlv1alpha1.ProxyList
|
||||
if err := s.Reader.List(r.Context(), &list); err != nil {
|
||||
s.log.Error(err, "listing proxies for lease")
|
||||
writeError(w, http.StatusInternalServerError, "internal", "listing proxies failed")
|
||||
return
|
||||
}
|
||||
|
||||
// The store gets only healthy, live candidates; unhealthy matches are
|
||||
// counted here because the store never sees them.
|
||||
var candidates []lease.Candidate
|
||||
byKey := map[string]*crawlv1alpha1.Proxy{}
|
||||
unhealthy := 0
|
||||
for i := range list.Items {
|
||||
p := &list.Items[i]
|
||||
if !p.DeletionTimestamp.IsZero() || !matchesAttributes(p, req.Selector) {
|
||||
continue
|
||||
}
|
||||
if !isHealthy(p) {
|
||||
unhealthy++
|
||||
continue
|
||||
}
|
||||
key := client.ObjectKeyFromObject(p).String()
|
||||
byKey[key] = p
|
||||
candidates = append(candidates, lease.Candidate{
|
||||
Proxy: key,
|
||||
MaxLeases: p.MaxLeasesOrDefault(),
|
||||
Latency: time.Duration(p.Status.LatencyMillis) * time.Millisecond,
|
||||
})
|
||||
}
|
||||
|
||||
granted, stats, err := s.Store.Acquire(r.Context(), lease.AcquireRequest{
|
||||
Candidates: candidates,
|
||||
Target: req.Target,
|
||||
TTL: ttl,
|
||||
})
|
||||
if errors.Is(err, lease.ErrNoMatch) {
|
||||
if s.Metrics != nil {
|
||||
s.Metrics.LeaseRequest("no_match")
|
||||
}
|
||||
writeJSON(w, http.StatusConflict, map[string]any{
|
||||
"error": "no_match",
|
||||
"message": "no healthy proxy with free capacity matched the selector",
|
||||
"considered": stats.Considered + unhealthy,
|
||||
"atCapacity": stats.AtCapacity,
|
||||
"inCooldown": stats.InCooldown,
|
||||
"unhealthy": unhealthy,
|
||||
})
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid_request", err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
if s.Metrics != nil {
|
||||
s.Metrics.LeaseRequest("granted")
|
||||
}
|
||||
writeJSON(w, http.StatusCreated, leaseResponse{
|
||||
LeaseID: granted.ID,
|
||||
Proxy: viewOf(byKey[granted.Proxy], s.Store.Counts()[granted.Proxy]),
|
||||
ExpiresAt: granted.ExpiresAt,
|
||||
TTLSeconds: int64(ttl.Seconds()),
|
||||
})
|
||||
}
|
||||
|
||||
// DELETE /v1/leases/{id} — early release, always 204: releasing an unknown
|
||||
// or already expired lease is not an error.
|
||||
func (s *Server) handleReleaseLease(w http.ResponseWriter, r *http.Request) {
|
||||
s.Store.Release(r.Context(), r.PathValue("id"))
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
|
||||
type reportRequest struct {
|
||||
Result string `json:"result"`
|
||||
Target string `json:"target"`
|
||||
}
|
||||
|
||||
// POST /v1/leases/{id}/report
|
||||
func (s *Server) handleReportLease(w http.ResponseWriter, r *http.Request) {
|
||||
var req reportRequest
|
||||
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
|
||||
writeError(w, http.StatusBadRequest, "invalid_body", err.Error())
|
||||
return
|
||||
}
|
||||
result, ok := lease.ParseResult(req.Result)
|
||||
if !ok {
|
||||
writeError(w, http.StatusBadRequest, "invalid_result",
|
||||
fmt.Sprintf("result must be one of ok, rate_limited, banned; got %q", req.Result))
|
||||
return
|
||||
}
|
||||
if err := s.Store.Report(r.Context(), r.PathValue("id"), result, req.Target); err != nil {
|
||||
if errors.Is(err, lease.ErrUnknownLease) {
|
||||
writeError(w, http.StatusNotFound, "unknown_lease", "no such lease")
|
||||
return
|
||||
}
|
||||
s.log.Error(err, "reporting lease", "leaseID", r.PathValue("id"))
|
||||
writeError(w, http.StatusInternalServerError, "internal", "report failed")
|
||||
return
|
||||
}
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
|
||||
func ptr[T any](v T) *T { return &v }
|
||||
234
internal/discovery/server.go
Normal file
234
internal/discovery/server.go
Normal file
@@ -0,0 +1,234 @@
|
||||
// Package discovery implements the HTTP API crawler clients use to find
|
||||
// and lease proxies: list healthy proxies filtered by attributes, acquire a
|
||||
// TTL-based lease, release it early, and report how a target treated the
|
||||
// proxy. Reads go through the manager's informer cache; lease state lives
|
||||
// in the injected store.
|
||||
package discovery
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/subtle"
|
||||
"encoding/json"
|
||||
"net"
|
||||
"net/http"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/go-logr/logr"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
logf "sigs.k8s.io/controller-runtime/pkg/log"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/lease"
|
||||
)
|
||||
|
||||
// LeaseStore is what the handlers need from a lease backend. Defined here,
|
||||
// consumer-side, so a CRD- or Redis-backed store can replace the in-memory
|
||||
// one (which internal/lease's *Store satisfies) without touching handlers.
|
||||
type LeaseStore interface {
|
||||
Acquire(ctx context.Context, req lease.AcquireRequest) (*lease.Lease, lease.AcquireStats, error)
|
||||
Release(ctx context.Context, id string)
|
||||
Report(ctx context.Context, id string, result lease.Result, target string) error
|
||||
Counts() map[string]int
|
||||
}
|
||||
|
||||
// LeaseMetrics counts lease acquisitions by outcome. Implemented by
|
||||
// internal/metrics; defined here so this package carries no metrics
|
||||
// dependency.
|
||||
type LeaseMetrics interface {
|
||||
LeaseRequest(outcome string)
|
||||
}
|
||||
|
||||
const (
|
||||
defaultAddr = ":8090"
|
||||
defaultTTL = 5 * time.Minute
|
||||
defaultMaxTTL = time.Hour
|
||||
maxBodyBytes = 64 << 10
|
||||
shutdownGrace = 10 * time.Second
|
||||
readHeadTimeout = 5 * time.Second
|
||||
)
|
||||
|
||||
// Server serves the discovery API as a manager Runnable.
|
||||
type Server struct {
|
||||
// Reader lists Proxies from the manager's cache.
|
||||
Reader client.Reader
|
||||
// Store is the lease backend.
|
||||
Store LeaseStore
|
||||
// Addr is the listen address (default ":8090"; --discovery-addr).
|
||||
Addr string
|
||||
// Token is the static bearer token from DISCOVERY_TOKEN. Empty
|
||||
// disables auth — allowed for the prototype, but loudly warned about
|
||||
// at startup, because in-cluster that is a silent security hole.
|
||||
Token string
|
||||
// MaxLeaseTTL caps requested lease TTLs (default 1h; --max-lease-ttl).
|
||||
MaxLeaseTTL time.Duration
|
||||
// Metrics, when non-nil, counts lease requests by outcome.
|
||||
Metrics LeaseMetrics
|
||||
|
||||
log logr.Logger
|
||||
|
||||
mu sync.Mutex
|
||||
boundAddr string
|
||||
}
|
||||
|
||||
// NeedLeaderElection is false, and the deployment ships replicas: 1.
|
||||
// Verified against controller-runtime's runnable ordering: caches start and
|
||||
// sync before non-leader-election runnables, so cache reads here are safe.
|
||||
// If this were leader-elected, non-leader replicas would refuse connections
|
||||
// while still being Service endpoints. The 1-replica constraint comes from
|
||||
// lease state being per-process — both facts are README caveats.
|
||||
func (s *Server) NeedLeaderElection() bool { return false }
|
||||
|
||||
// BoundAddr returns the actual listen address once Start has bound it —
|
||||
// meaningful when Addr uses port 0 (tests).
|
||||
func (s *Server) BoundAddr() string {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
return s.boundAddr
|
||||
}
|
||||
|
||||
// Start listens and serves until ctx ends, then shuts down gracefully with
|
||||
// a 10-second grace period.
|
||||
func (s *Server) Start(ctx context.Context) error {
|
||||
if s.Addr == "" {
|
||||
s.Addr = defaultAddr
|
||||
}
|
||||
if s.MaxLeaseTTL == 0 {
|
||||
s.MaxLeaseTTL = defaultMaxTTL
|
||||
}
|
||||
s.log = logf.FromContext(ctx).WithName("discovery")
|
||||
if s.Token == "" {
|
||||
s.log.Info("WARNING: DISCOVERY_TOKEN is empty — the discovery API is served without authentication")
|
||||
}
|
||||
|
||||
ln, err := net.Listen("tcp", s.Addr)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
s.mu.Lock()
|
||||
s.boundAddr = ln.Addr().String()
|
||||
s.mu.Unlock()
|
||||
|
||||
srv := &http.Server{
|
||||
Handler: s.handler(),
|
||||
ReadHeaderTimeout: readHeadTimeout,
|
||||
ReadTimeout: 10 * time.Second,
|
||||
WriteTimeout: 10 * time.Second,
|
||||
IdleTimeout: 60 * time.Second,
|
||||
}
|
||||
|
||||
errCh := make(chan error, 1)
|
||||
go func() { errCh <- srv.Serve(ln) }()
|
||||
s.log.Info("discovery API listening", "addr", s.boundAddr)
|
||||
|
||||
select {
|
||||
case err := <-errCh:
|
||||
return err
|
||||
case <-ctx.Done():
|
||||
shutdownCtx, cancel := context.WithTimeout(context.Background(), shutdownGrace)
|
||||
defer cancel()
|
||||
if err := srv.Shutdown(shutdownCtx); err != nil {
|
||||
return err
|
||||
}
|
||||
<-errCh // always http.ErrServerClosed after a clean Shutdown
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// handler assembles the mux and the middleware chain, outermost first:
|
||||
// recover → request-log → body-size cap → bearer auth.
|
||||
func (s *Server) handler() http.Handler {
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("GET /healthz", func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
_, _ = w.Write([]byte("ok\n"))
|
||||
})
|
||||
mux.HandleFunc("GET /v1/proxies", s.handleListProxies)
|
||||
mux.HandleFunc("POST /v1/leases", s.handleAcquireLease)
|
||||
mux.HandleFunc("DELETE /v1/leases/{id}", s.handleReleaseLease)
|
||||
mux.HandleFunc("POST /v1/leases/{id}/report", s.handleReportLease)
|
||||
|
||||
var h http.Handler = mux
|
||||
h = s.authMiddleware(h)
|
||||
h = maxBytesMiddleware(h)
|
||||
h = s.logMiddleware(h)
|
||||
h = s.recoverMiddleware(h)
|
||||
return h
|
||||
}
|
||||
|
||||
func (s *Server) recoverMiddleware(next http.Handler) http.Handler {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
defer func() {
|
||||
if p := recover(); p != nil {
|
||||
s.log.Error(nil, "panic in discovery handler", "panic", p, "path", r.URL.Path)
|
||||
writeError(w, http.StatusInternalServerError, "internal", "internal server error")
|
||||
}
|
||||
}()
|
||||
next.ServeHTTP(w, r)
|
||||
})
|
||||
}
|
||||
|
||||
// statusRecorder captures the response code for the request log.
|
||||
type statusRecorder struct {
|
||||
http.ResponseWriter
|
||||
status int
|
||||
}
|
||||
|
||||
func (r *statusRecorder) WriteHeader(code int) {
|
||||
r.status = code
|
||||
r.ResponseWriter.WriteHeader(code)
|
||||
}
|
||||
|
||||
func (s *Server) logMiddleware(next http.Handler) http.Handler {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.URL.Path == "/healthz" {
|
||||
next.ServeHTTP(w, r) // probes are noise
|
||||
return
|
||||
}
|
||||
rec := &statusRecorder{ResponseWriter: w, status: http.StatusOK}
|
||||
start := time.Now()
|
||||
next.ServeHTTP(rec, r)
|
||||
s.log.Info("request",
|
||||
"method", r.Method, "path", r.URL.Path,
|
||||
"status", rec.status, "duration", time.Since(start).String())
|
||||
})
|
||||
}
|
||||
|
||||
func maxBytesMiddleware(next http.Handler) http.Handler {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
r.Body = http.MaxBytesReader(w, r.Body, maxBodyBytes)
|
||||
next.ServeHTTP(w, r)
|
||||
})
|
||||
}
|
||||
|
||||
func (s *Server) authMiddleware(next http.Handler) http.Handler {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if s.Token == "" || r.URL.Path == "/healthz" {
|
||||
next.ServeHTTP(w, r)
|
||||
return
|
||||
}
|
||||
token, ok := strings.CutPrefix(r.Header.Get("Authorization"), "Bearer ")
|
||||
if !ok || subtle.ConstantTimeCompare([]byte(token), []byte(s.Token)) != 1 {
|
||||
writeError(w, http.StatusUnauthorized, "unauthorized", "missing or invalid bearer token")
|
||||
return
|
||||
}
|
||||
next.ServeHTTP(w, r)
|
||||
})
|
||||
}
|
||||
|
||||
// errorBody is the shared error shape:
|
||||
// {"error":"<machine_code>","message":"<human>"}.
|
||||
type errorBody struct {
|
||||
Error string `json:"error"`
|
||||
Message string `json:"message"`
|
||||
}
|
||||
|
||||
func writeError(w http.ResponseWriter, status int, code, message string) {
|
||||
writeJSON(w, status, errorBody{Error: code, Message: message})
|
||||
}
|
||||
|
||||
func writeJSON(w http.ResponseWriter, status int, v any) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
w.WriteHeader(status)
|
||||
_ = json.NewEncoder(w).Encode(v)
|
||||
}
|
||||
386
internal/discovery/server_test.go
Normal file
386
internal/discovery/server_test.go
Normal file
@@ -0,0 +1,386 @@
|
||||
package discovery
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/lease"
|
||||
)
|
||||
|
||||
func testProxy(name string, attrs map[string]string, healthy bool, mut ...func(*crawlv1alpha1.Proxy)) *crawlv1alpha1.Proxy {
|
||||
p := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: name, Namespace: "default", UID: types.UID("uid-" + name),
|
||||
},
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "10.0.0.1", Port: 3128},
|
||||
Attributes: attrs,
|
||||
},
|
||||
}
|
||||
p.Status.IP = "10.0.0.1"
|
||||
p.Status.Phase = crawlv1alpha1.PhaseReady
|
||||
status := metav1.ConditionFalse
|
||||
if healthy {
|
||||
status = metav1.ConditionTrue
|
||||
}
|
||||
p.Status.Conditions = []metav1.Condition{{
|
||||
Type: crawlv1alpha1.ConditionHealthy, Status: status,
|
||||
Reason: "Probing", LastTransitionTime: metav1.Now(),
|
||||
}}
|
||||
for _, m := range mut {
|
||||
m(p)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
func withMaxLeases(n int32) func(*crawlv1alpha1.Proxy) {
|
||||
return func(p *crawlv1alpha1.Proxy) { p.Spec.MaxLeases = &n }
|
||||
}
|
||||
|
||||
func withLatency(ms int64) func(*crawlv1alpha1.Proxy) {
|
||||
return func(p *crawlv1alpha1.Proxy) { p.Status.LatencyMillis = ms }
|
||||
}
|
||||
|
||||
// newTestServer wires the handler chain to a fake cache reader and a real
|
||||
// lease store, served over httptest.
|
||||
func newTestServer(t *testing.T, token string, proxies ...*crawlv1alpha1.Proxy) (*httptest.Server, *Server) {
|
||||
t.Helper()
|
||||
s := runtime.NewScheme()
|
||||
if err := crawlv1alpha1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("scheme: %v", err)
|
||||
}
|
||||
builder := fake.NewClientBuilder().WithScheme(s)
|
||||
for _, p := range proxies {
|
||||
builder = builder.WithObjects(p)
|
||||
}
|
||||
srv := &Server{
|
||||
Reader: builder.Build(),
|
||||
Store: lease.NewStore(15 * time.Minute),
|
||||
Token: token,
|
||||
MaxLeaseTTL: time.Hour,
|
||||
}
|
||||
ts := httptest.NewServer(srv.handler())
|
||||
t.Cleanup(ts.Close)
|
||||
return ts, srv
|
||||
}
|
||||
|
||||
type response struct {
|
||||
status int
|
||||
body map[string]any
|
||||
}
|
||||
|
||||
func do(t *testing.T, ts *httptest.Server, method, path, token string, body any) response {
|
||||
t.Helper()
|
||||
var reader io.Reader
|
||||
if body != nil {
|
||||
if s, ok := body.(string); ok {
|
||||
reader = bytes.NewBufferString(s)
|
||||
} else {
|
||||
b, err := json.Marshal(body)
|
||||
if err != nil {
|
||||
t.Fatalf("marshaling request body: %v", err)
|
||||
}
|
||||
reader = bytes.NewBuffer(b)
|
||||
}
|
||||
}
|
||||
req, err := http.NewRequestWithContext(context.Background(), method, ts.URL+path, reader)
|
||||
if err != nil {
|
||||
t.Fatalf("building request: %v", err)
|
||||
}
|
||||
if token != "" {
|
||||
req.Header.Set("Authorization", "Bearer "+token)
|
||||
}
|
||||
resp, err := ts.Client().Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("%s %s: %v", method, path, err)
|
||||
}
|
||||
defer func() { _ = resp.Body.Close() }()
|
||||
out := response{status: resp.StatusCode}
|
||||
raw, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
t.Fatalf("reading response: %v", err)
|
||||
}
|
||||
if len(raw) > 0 && resp.Header.Get("Content-Type") == "application/json" {
|
||||
if err := json.Unmarshal(raw, &out.body); err != nil {
|
||||
t.Fatalf("decoding response %q: %v", raw, err)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func TestAuth(t *testing.T) {
|
||||
t.Parallel()
|
||||
ts, _ := newTestServer(t, "sekrit", testProxy("p1", nil, true))
|
||||
|
||||
if got := do(t, ts, http.MethodGet, "/v1/proxies", "", nil); got.status != http.StatusUnauthorized {
|
||||
t.Errorf("no token: status %d, want 401", got.status)
|
||||
}
|
||||
if got := do(t, ts, http.MethodGet, "/v1/proxies", "wrong", nil); got.status != http.StatusUnauthorized {
|
||||
t.Errorf("wrong token: status %d, want 401", got.status)
|
||||
}
|
||||
if got := do(t, ts, http.MethodGet, "/v1/proxies", "sekrit", nil); got.status != http.StatusOK {
|
||||
t.Errorf("correct token: status %d, want 200", got.status)
|
||||
}
|
||||
if got := do(t, ts, http.MethodGet, "/healthz", "", nil); got.status != http.StatusOK {
|
||||
t.Errorf("healthz without token: status %d, want 200 (always unauthenticated)", got.status)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAuth_disabledWithEmptyToken(t *testing.T) {
|
||||
t.Parallel()
|
||||
ts, _ := newTestServer(t, "", testProxy("p1", nil, true))
|
||||
if got := do(t, ts, http.MethodGet, "/v1/proxies", "", nil); got.status != http.StatusOK {
|
||||
t.Errorf("status %d, want 200 with auth disabled", got.status)
|
||||
}
|
||||
}
|
||||
|
||||
func TestListProxies(t *testing.T) {
|
||||
t.Parallel()
|
||||
ts, _ := newTestServer(t, "",
|
||||
testProxy("eu-healthy", map[string]string{"geo": "eu", "purpose": "crawl"}, true),
|
||||
testProxy("eu-sick", map[string]string{"geo": "eu"}, false),
|
||||
testProxy("us-healthy", map[string]string{"geo": "us"}, true),
|
||||
)
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
query string
|
||||
wantCount int
|
||||
wantFirst string
|
||||
}{
|
||||
{name: "no filter returns everything", query: "", wantCount: 3, wantFirst: "default/eu-healthy"},
|
||||
{name: "healthy filter", query: "?healthy=true", wantCount: 2},
|
||||
{name: "unhealthy filter", query: "?healthy=false", wantCount: 1, wantFirst: "default/eu-sick"},
|
||||
{name: "attribute filter", query: "?attr.geo=eu", wantCount: 2},
|
||||
{name: "attribute and health combined", query: "?attr.geo=eu&healthy=true", wantCount: 1, wantFirst: "default/eu-healthy"},
|
||||
{name: "two attributes must both match", query: "?attr.geo=eu&attr.purpose=crawl", wantCount: 1, wantFirst: "default/eu-healthy"},
|
||||
{name: "no matches is 200 with count 0", query: "?attr.geo=mars", wantCount: 0},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
got := do(t, ts, http.MethodGet, "/v1/proxies"+tc.query, "", nil)
|
||||
if got.status != http.StatusOK {
|
||||
t.Fatalf("status %d, want 200", got.status)
|
||||
}
|
||||
count := int(got.body["count"].(float64))
|
||||
proxies := got.body["proxies"].([]any)
|
||||
if count != tc.wantCount || len(proxies) != tc.wantCount {
|
||||
t.Fatalf("count = %d (len %d), want %d", count, len(proxies), tc.wantCount)
|
||||
}
|
||||
if tc.wantFirst != "" {
|
||||
first := proxies[0].(map[string]any)
|
||||
if first["id"] != tc.wantFirst {
|
||||
t.Errorf("first id = %v, want %s", first["id"], tc.wantFirst)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
t.Run("invalid healthy value is 400", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
if got := do(t, ts, http.MethodGet, "/v1/proxies?healthy=maybe", "", nil); got.status != http.StatusBadRequest {
|
||||
t.Errorf("status %d, want 400", got.status)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestAcquireLease_grantShape(t *testing.T) {
|
||||
t.Parallel()
|
||||
ts, _ := newTestServer(t, "",
|
||||
testProxy("eu1", map[string]string{"geo": "eu"}, true, withLatency(30)),
|
||||
testProxy("eu2", map[string]string{"geo": "eu"}, true, withLatency(10)),
|
||||
)
|
||||
|
||||
got := do(t, ts, http.MethodPost, "/v1/leases", "", map[string]any{
|
||||
"selector": map[string]string{"geo": "eu"},
|
||||
})
|
||||
if got.status != http.StatusCreated {
|
||||
t.Fatalf("status %d (%v), want 201", got.status, got.body)
|
||||
}
|
||||
if got.body["leaseID"] == "" || got.body["leaseID"] == nil {
|
||||
t.Error("empty leaseID")
|
||||
}
|
||||
if got.body["ttlSeconds"].(float64) != 300 {
|
||||
t.Errorf("ttlSeconds = %v, want the 300 default", got.body["ttlSeconds"])
|
||||
}
|
||||
proxy := got.body["proxy"].(map[string]any)
|
||||
if proxy["id"] != "default/eu2" {
|
||||
t.Errorf("granted %v, want default/eu2 (lower latency at equal load)", proxy["id"])
|
||||
}
|
||||
if proxy["activeLeases"].(float64) != 1 {
|
||||
t.Errorf("activeLeases = %v, want 1 (this grant included)", proxy["activeLeases"])
|
||||
}
|
||||
if _, err := time.Parse(time.RFC3339, got.body["expiresAt"].(string)); err != nil {
|
||||
t.Errorf("expiresAt %v is not RFC3339: %v", got.body["expiresAt"], err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAcquireLease_noMatchBody(t *testing.T) {
|
||||
t.Parallel()
|
||||
ts, _ := newTestServer(t, "",
|
||||
testProxy("eu-tiny", map[string]string{"geo": "eu"}, true, withMaxLeases(1)),
|
||||
testProxy("eu-sick", map[string]string{"geo": "eu"}, false),
|
||||
)
|
||||
|
||||
body := map[string]any{"selector": map[string]string{"geo": "eu"}}
|
||||
if got := do(t, ts, http.MethodPost, "/v1/leases", "", body); got.status != http.StatusCreated {
|
||||
t.Fatalf("first acquire: status %d, want 201", got.status)
|
||||
}
|
||||
|
||||
got := do(t, ts, http.MethodPost, "/v1/leases", "", body)
|
||||
if got.status != http.StatusConflict {
|
||||
t.Fatalf("second acquire: status %d, want 409", got.status)
|
||||
}
|
||||
want := map[string]float64{"considered": 2, "atCapacity": 1, "inCooldown": 0, "unhealthy": 1}
|
||||
for k, v := range want {
|
||||
if got.body[k].(float64) != v {
|
||||
t.Errorf("%s = %v, want %v (body %v)", k, got.body[k], v, got.body)
|
||||
}
|
||||
}
|
||||
if got.body["error"] != "no_match" {
|
||||
t.Errorf("error = %v, want no_match", got.body["error"])
|
||||
}
|
||||
}
|
||||
|
||||
func TestAcquireLease_badRequests(t *testing.T) {
|
||||
t.Parallel()
|
||||
ts, _ := newTestServer(t, "", testProxy("p1", nil, true))
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
body any
|
||||
wantCode string
|
||||
}{
|
||||
{name: "ttl above the cap", body: map[string]any{"ttlSeconds": 999999}, wantCode: "invalid_ttl"},
|
||||
{name: "negative ttl", body: map[string]any{"ttlSeconds": -5}, wantCode: "invalid_ttl"},
|
||||
{name: "malformed json", body: "{not json", wantCode: "invalid_body"},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
got := do(t, ts, http.MethodPost, "/v1/leases", "", tc.body)
|
||||
if got.status != http.StatusBadRequest || got.body["error"] != tc.wantCode {
|
||||
t.Errorf("= %d/%v, want 400/%s", got.status, got.body["error"], tc.wantCode)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestReleaseLease_alwaysNoContent(t *testing.T) {
|
||||
t.Parallel()
|
||||
ts, _ := newTestServer(t, "", testProxy("p1", nil, true))
|
||||
|
||||
got := do(t, ts, http.MethodPost, "/v1/leases", "", map[string]any{})
|
||||
if got.status != http.StatusCreated {
|
||||
t.Fatalf("acquire: status %d, want 201", got.status)
|
||||
}
|
||||
id := got.body["leaseID"].(string)
|
||||
|
||||
for _, path := range []string{"/v1/leases/" + id, "/v1/leases/" + id, "/v1/leases/never-existed"} {
|
||||
if got := do(t, ts, http.MethodDelete, path, "", nil); got.status != http.StatusNoContent {
|
||||
t.Errorf("DELETE %s: status %d, want 204", path, got.status)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestReportLease(t *testing.T) {
|
||||
t.Parallel()
|
||||
ts, _ := newTestServer(t, "", testProxy("p1", map[string]string{"geo": "eu"}, true))
|
||||
|
||||
got := do(t, ts, http.MethodPost, "/v1/leases", "", map[string]any{
|
||||
"selector": map[string]string{"geo": "eu"}, "target": "example.com",
|
||||
})
|
||||
if got.status != http.StatusCreated {
|
||||
t.Fatalf("acquire: status %d, want 201", got.status)
|
||||
}
|
||||
id := got.body["leaseID"].(string)
|
||||
reportPath := fmt.Sprintf("/v1/leases/%s/report", id)
|
||||
|
||||
if got := do(t, ts, http.MethodPost, reportPath, "", map[string]any{"result": "rate_limited", "target": "example.com"}); got.status != http.StatusNoContent {
|
||||
t.Fatalf("report: status %d, want 204", got.status)
|
||||
}
|
||||
|
||||
// The cooldown from the report now blocks same-target acquisition.
|
||||
got = do(t, ts, http.MethodPost, "/v1/leases", "", map[string]any{
|
||||
"selector": map[string]string{"geo": "eu"}, "target": "example.com",
|
||||
})
|
||||
if got.status != http.StatusConflict || got.body["inCooldown"].(float64) != 1 {
|
||||
t.Errorf("post-report acquire = %d/%v, want 409 with inCooldown 1", got.status, got.body)
|
||||
}
|
||||
|
||||
t.Run("invalid result value", func(t *testing.T) {
|
||||
got := do(t, ts, http.MethodPost, reportPath, "", map[string]any{"result": "throttled"})
|
||||
if got.status != http.StatusBadRequest || got.body["error"] != "invalid_result" {
|
||||
t.Errorf("= %d/%v, want 400/invalid_result", got.status, got.body["error"])
|
||||
}
|
||||
})
|
||||
t.Run("unknown lease", func(t *testing.T) {
|
||||
got := do(t, ts, http.MethodPost, "/v1/leases/never-existed/report", "", map[string]any{"result": "ok"})
|
||||
if got.status != http.StatusNotFound || got.body["error"] != "unknown_lease" {
|
||||
t.Errorf("= %d/%v, want 404/unknown_lease", got.status, got.body["error"])
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestStart_servesAndShutsDown(t *testing.T) {
|
||||
t.Parallel()
|
||||
s := runtime.NewScheme()
|
||||
if err := crawlv1alpha1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("scheme: %v", err)
|
||||
}
|
||||
srv := &Server{
|
||||
Reader: fake.NewClientBuilder().WithScheme(s).Build(),
|
||||
Store: lease.NewStore(time.Minute),
|
||||
Addr: "127.0.0.1:0",
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- srv.Start(ctx) }()
|
||||
|
||||
var addr string
|
||||
deadline := time.After(5 * time.Second)
|
||||
for addr == "" {
|
||||
select {
|
||||
case <-deadline:
|
||||
t.Fatal("server never bound")
|
||||
case <-time.After(5 * time.Millisecond):
|
||||
addr = srv.BoundAddr()
|
||||
}
|
||||
}
|
||||
|
||||
resp, err := http.Get("http://" + addr + "/healthz")
|
||||
if err != nil {
|
||||
t.Fatalf("healthz: %v", err)
|
||||
}
|
||||
_ = resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("healthz status %d, want 200", resp.StatusCode)
|
||||
}
|
||||
|
||||
cancel()
|
||||
select {
|
||||
case err := <-done:
|
||||
if err != nil {
|
||||
t.Errorf("Start returned %v, want nil after graceful shutdown", err)
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("Start did not stop on cancel")
|
||||
}
|
||||
}
|
||||
125
internal/gc/gc.go
Normal file
125
internal/gc/gc.go
Normal file
@@ -0,0 +1,125 @@
|
||||
// Package gc implements orphan garbage collection: a periodic sweep that
|
||||
// deletes provider instances tagged by this operator whose owning Proxy CR
|
||||
// no longer exists — the safety net for crashes between a provider Create
|
||||
// and the status write that records it.
|
||||
package gc
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"time"
|
||||
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
logf "sigs.k8s.io/controller-runtime/pkg/log"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// Sweeper is the manager Runnable running the sweep loop.
|
||||
type Sweeper struct {
|
||||
// Reader lists Proxies from the manager's cache to establish the live
|
||||
// UID set.
|
||||
Reader client.Reader
|
||||
// Providers are the configured backends; each is swept independently.
|
||||
Providers map[string]provider.Provider
|
||||
|
||||
// Interval between sweeps (default 10m). The first sweep runs one full
|
||||
// interval after start, not immediately — right after startup the
|
||||
// cache is coldest and an in-flight create is most likely.
|
||||
Interval time.Duration
|
||||
// MinAge exempts young instances (default 10m): an instance mid-create
|
||||
// may not have its status write landed yet; deleting it would race the
|
||||
// reconciler.
|
||||
MinAge time.Duration
|
||||
|
||||
// NamespaceRestricted must be set when the manager cache is limited to
|
||||
// one namespace. Then the live-UID set is incomplete, and a sweep
|
||||
// would delete VMs owned by Proxies the cache cannot see — so Start
|
||||
// refuses unless AllowNamespaced (--gc-allow-namespaced) is explicit.
|
||||
NamespaceRestricted bool
|
||||
AllowNamespaced bool
|
||||
|
||||
now func() time.Time
|
||||
}
|
||||
|
||||
// NeedLeaderElection is true: the sweep is destructive and must have a
|
||||
// single writer.
|
||||
func (s *Sweeper) NeedLeaderElection() bool { return true }
|
||||
|
||||
// Start runs the sweep loop until ctx ends.
|
||||
func (s *Sweeper) Start(ctx context.Context) error {
|
||||
if s.NamespaceRestricted && !s.AllowNamespaced {
|
||||
return errors.New(
|
||||
"orphan GC refuses to run against a namespace-restricted cache: proxies outside the namespace " +
|
||||
"would count as orphans and their instances would be deleted; pass --gc-allow-namespaced to override")
|
||||
}
|
||||
if s.Interval == 0 {
|
||||
s.Interval = 10 * time.Minute
|
||||
}
|
||||
if s.MinAge == 0 {
|
||||
s.MinAge = 10 * time.Minute
|
||||
}
|
||||
if s.now == nil {
|
||||
s.now = time.Now
|
||||
}
|
||||
|
||||
ticker := time.NewTicker(s.Interval)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return nil
|
||||
case <-ticker.C:
|
||||
s.sweep(ctx)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// sweep deletes tagged instances whose UID matches no existing Proxy CR.
|
||||
// A CR with a deletionTimestamp still counts as live: its finalizer owns
|
||||
// that deletion, and GC racing it would double-delete. A UID becomes
|
||||
// orphan-eligible only once the object is fully gone.
|
||||
func (s *Sweeper) sweep(ctx context.Context) {
|
||||
log := logf.FromContext(ctx).WithName("orphan-gc")
|
||||
|
||||
var list crawlv1alpha1.ProxyList
|
||||
if err := s.Reader.List(ctx, &list); err != nil {
|
||||
// Without the live set nothing can be proven orphaned; skip the
|
||||
// whole sweep rather than guess.
|
||||
log.Error(err, "listing proxies; skipping this sweep")
|
||||
return
|
||||
}
|
||||
live := make(map[string]bool, len(list.Items))
|
||||
for i := range list.Items {
|
||||
live[string(list.Items[i].UID)] = true
|
||||
}
|
||||
|
||||
for name, prov := range s.Providers {
|
||||
instances, err := prov.ListByTag(ctx)
|
||||
if err != nil {
|
||||
// One broken provider must not abort the sweep for the rest.
|
||||
log.Error(err, "listing instances; skipping this provider", "provider", name)
|
||||
continue
|
||||
}
|
||||
for _, inst := range instances {
|
||||
switch {
|
||||
case inst.UID == "":
|
||||
// Managed label without a UID label shouldn't exist for
|
||||
// anything this operator created; without ownership proof,
|
||||
// never delete.
|
||||
continue
|
||||
case live[inst.UID]:
|
||||
continue
|
||||
case s.now().Sub(inst.CreatedAt) < s.MinAge:
|
||||
continue
|
||||
}
|
||||
log.Info("WARNING: deleting orphaned instance",
|
||||
"provider", name, "providerID", inst.ID, "uid", inst.UID)
|
||||
if err := prov.Delete(ctx, inst.ID); err != nil {
|
||||
log.Error(err, "deleting orphaned instance",
|
||||
"provider", name, "providerID", inst.ID, "uid", inst.UID)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
214
internal/gc/gc_test.go
Normal file
214
internal/gc/gc_test.go
Normal file
@@ -0,0 +1,214 @@
|
||||
package gc
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// listProvider serves a canned instance list and records deletions.
|
||||
type listProvider struct {
|
||||
mu sync.Mutex
|
||||
instances []provider.Instance
|
||||
listErr error
|
||||
deleted []string
|
||||
}
|
||||
|
||||
func (l *listProvider) Create(context.Context, provider.CreateRequest) (string, error) {
|
||||
return "", errors.New("not used")
|
||||
}
|
||||
|
||||
func (l *listProvider) Get(context.Context, string) (*provider.Instance, error) {
|
||||
return nil, provider.ErrNotFound
|
||||
}
|
||||
|
||||
func (l *listProvider) Delete(_ context.Context, id string) error {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
l.deleted = append(l.deleted, id)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (l *listProvider) ListByTag(context.Context) ([]provider.Instance, error) {
|
||||
return l.instances, l.listErr
|
||||
}
|
||||
|
||||
func (l *listProvider) deletedIDs() []string {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
return append([]string(nil), l.deleted...)
|
||||
}
|
||||
|
||||
func proxyWithUID(name, uid string, mut ...func(*crawlv1alpha1.Proxy)) *crawlv1alpha1.Proxy {
|
||||
p := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: "default", UID: types.UID(uid)},
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged,
|
||||
Provider: "stub",
|
||||
},
|
||||
}
|
||||
for _, m := range mut {
|
||||
m(p)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
func newReader(t *testing.T, objs ...client.Object) client.Reader {
|
||||
t.Helper()
|
||||
s := runtime.NewScheme()
|
||||
if err := crawlv1alpha1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("scheme: %v", err)
|
||||
}
|
||||
return fake.NewClientBuilder().WithScheme(s).WithObjects(objs...).Build()
|
||||
}
|
||||
|
||||
func oldInstance(id, uid string) provider.Instance {
|
||||
return provider.Instance{ID: id, UID: uid, State: provider.StateRunning,
|
||||
CreatedAt: time.Now().Add(-time.Hour)}
|
||||
}
|
||||
|
||||
func newSweeper(reader client.Reader, providers map[string]provider.Provider) *Sweeper {
|
||||
return &Sweeper{
|
||||
Reader: reader,
|
||||
Providers: providers,
|
||||
Interval: 10 * time.Minute,
|
||||
MinAge: 10 * time.Minute,
|
||||
now: time.Now,
|
||||
}
|
||||
}
|
||||
|
||||
func TestSweep_deletesOnlyTrueOrphans(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
deletingCR := proxyWithUID("deleting", "uid-deleting", func(p *crawlv1alpha1.Proxy) {
|
||||
now := metav1.Now()
|
||||
p.DeletionTimestamp = &now
|
||||
p.Finalizers = []string{crawlv1alpha1.FinalizerName}
|
||||
})
|
||||
prov := &listProvider{instances: []provider.Instance{
|
||||
oldInstance("inst-live", "uid-live"),
|
||||
oldInstance("inst-orphan", "uid-orphan"),
|
||||
oldInstance("inst-deleting", "uid-deleting"),
|
||||
{ID: "inst-young", UID: "uid-young-orphan", State: provider.StateRunning,
|
||||
CreatedAt: time.Now().Add(-time.Minute)},
|
||||
oldInstance("inst-unlabelled", ""),
|
||||
}}
|
||||
s := newSweeper(
|
||||
newReader(t, proxyWithUID("live", "uid-live"), deletingCR),
|
||||
map[string]provider.Provider{"stub": prov},
|
||||
)
|
||||
|
||||
s.sweep(context.Background())
|
||||
|
||||
got := prov.deletedIDs()
|
||||
if len(got) != 1 || got[0] != "inst-orphan" {
|
||||
t.Errorf("deleted %v, want exactly [inst-orphan]:\n"+
|
||||
"live CR's instance must stay; a deleting CR still owns its instance (finalizer, not GC);\n"+
|
||||
"young instances may be mid-create; unlabelled instances have no ownership proof", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSweep_providerErrorDoesNotAbortOthers(t *testing.T) {
|
||||
t.Parallel()
|
||||
broken := &listProvider{listErr: errors.New("cloud is down")}
|
||||
working := &listProvider{instances: []provider.Instance{oldInstance("inst-orphan", "uid-orphan")}}
|
||||
s := newSweeper(newReader(t), map[string]provider.Provider{
|
||||
"broken": broken,
|
||||
"working": working,
|
||||
})
|
||||
|
||||
s.sweep(context.Background())
|
||||
|
||||
if got := working.deletedIDs(); len(got) != 1 {
|
||||
t.Errorf("working provider deletions = %v, want the orphan despite the broken provider", got)
|
||||
}
|
||||
}
|
||||
|
||||
// errReader fails every List: without the live set nothing can be proven
|
||||
// orphaned, so the sweep must delete nothing.
|
||||
type errReader struct{ client.Reader }
|
||||
|
||||
func (errReader) List(context.Context, client.ObjectList, ...client.ListOption) error {
|
||||
return errors.New("cache broken")
|
||||
}
|
||||
|
||||
func TestSweep_listFailureSkipsSweep(t *testing.T) {
|
||||
t.Parallel()
|
||||
prov := &listProvider{instances: []provider.Instance{oldInstance("inst-orphan", "uid-orphan")}}
|
||||
s := newSweeper(errReader{}, map[string]provider.Provider{"stub": prov})
|
||||
|
||||
s.sweep(context.Background())
|
||||
|
||||
if got := prov.deletedIDs(); len(got) != 0 {
|
||||
t.Errorf("deleted %v with an unreadable live set, want nothing", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStart_namespaceGuard(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
s := newSweeper(newReader(t), nil)
|
||||
s.NamespaceRestricted = true
|
||||
err := s.Start(context.Background())
|
||||
if err == nil || !strings.Contains(err.Error(), "--gc-allow-namespaced") {
|
||||
t.Errorf("Start with restricted cache = %v, want refusal naming the override flag", err)
|
||||
}
|
||||
|
||||
s2 := newSweeper(newReader(t), nil)
|
||||
s2.NamespaceRestricted = true
|
||||
s2.AllowNamespaced = true
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- s2.Start(ctx) }()
|
||||
cancel()
|
||||
select {
|
||||
case err := <-done:
|
||||
if err != nil {
|
||||
t.Errorf("Start with override = %v, want it to run until cancel", err)
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("Start did not stop on cancel")
|
||||
}
|
||||
}
|
||||
|
||||
func TestStart_sweepsOnIntervalAndStops(t *testing.T) {
|
||||
t.Parallel()
|
||||
prov := &listProvider{instances: []provider.Instance{oldInstance("inst-orphan", "uid-orphan")}}
|
||||
s := newSweeper(newReader(t), map[string]provider.Provider{"stub": prov})
|
||||
s.Interval = 5 * time.Millisecond
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- s.Start(ctx) }()
|
||||
|
||||
deadline := time.After(5 * time.Second)
|
||||
for len(prov.deletedIDs()) == 0 {
|
||||
select {
|
||||
case <-deadline:
|
||||
t.Fatal("no sweep ran")
|
||||
case <-time.After(2 * time.Millisecond):
|
||||
}
|
||||
}
|
||||
|
||||
cancel()
|
||||
select {
|
||||
case err := <-done:
|
||||
if err != nil {
|
||||
t.Errorf("Start = %v, want nil", err)
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("Start did not stop on cancel")
|
||||
}
|
||||
}
|
||||
361
internal/health/engine.go
Normal file
361
internal/health/engine.go
Normal file
@@ -0,0 +1,361 @@
|
||||
package health
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"math/rand/v2"
|
||||
"net"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/event"
|
||||
logf "sigs.k8s.io/controller-runtime/pkg/log"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
// Snapshot is the engine's current verdict for one proxy, read by the
|
||||
// reconciler when it represents health in the Proxy's status.
|
||||
type Snapshot struct {
|
||||
Healthy bool
|
||||
// Latency is the wall time of the most recent successful probe.
|
||||
Latency time.Duration
|
||||
// LastProbe is when the most recent probe (of either outcome) finished.
|
||||
LastProbe time.Time
|
||||
// LastError is the most recent probe failure; empty after a success.
|
||||
LastError string
|
||||
// ConsecutiveFailures is the current failure streak.
|
||||
ConsecutiveFailures int32
|
||||
}
|
||||
|
||||
// state is the engine's threshold bookkeeping for one proxy. The reported*
|
||||
// fields track what has been delivered over Events; they only advance when a
|
||||
// send succeeds, so a dropped event is retried after the next probe.
|
||||
type state struct {
|
||||
uid types.UID
|
||||
inFlight bool
|
||||
nextDue time.Time
|
||||
|
||||
healthy *bool // nil until a first verdict exists
|
||||
consecOK int32
|
||||
consecFail int32
|
||||
latency time.Duration
|
||||
lastProbe time.Time
|
||||
lastErr string
|
||||
|
||||
reportedHealthy *bool
|
||||
reportedLatency time.Duration
|
||||
lastReport time.Time
|
||||
}
|
||||
|
||||
type probeJob struct {
|
||||
key types.NamespacedName
|
||||
uid types.UID
|
||||
proxyURL *url.URL
|
||||
hc crawlv1alpha1.HealthCheckSpec
|
||||
interval time.Duration
|
||||
}
|
||||
|
||||
// ProbeMetrics receives every probe result and the retirement of a
|
||||
// proxy's series. Implemented by internal/metrics; defined here so this
|
||||
// package carries no metrics dependency.
|
||||
type ProbeMetrics interface {
|
||||
ObserveProbe(proxy string, latency time.Duration, success bool)
|
||||
ForgetProxy(proxy string)
|
||||
}
|
||||
|
||||
// Engine runs the probe scheduler and worker pool as a manager Runnable. It
|
||||
// never writes Proxy status itself — keeping the reconciler the single
|
||||
// status writer — and instead emits a GenericEvent per status-affecting
|
||||
// transition, which the reconciler consumes via source.Channel.
|
||||
type Engine struct {
|
||||
// Reader lists Proxies from the manager's cache each tick.
|
||||
Reader client.Reader
|
||||
// Events carries one enqueue-request per status-affecting transition.
|
||||
// Sends are non-blocking: a wedged reconciler must never stall probing.
|
||||
Events chan event.GenericEvent
|
||||
|
||||
// Workers is the probe worker pool size (default 8).
|
||||
Workers int
|
||||
// Tick is the scheduler interval (default 1s). At tens of proxies a
|
||||
// per-second list scan is free; a timer wheel would be unjustified.
|
||||
Tick time.Duration
|
||||
// MinReportInterval rate-limits latency-only status reports (default 60s).
|
||||
MinReportInterval time.Duration
|
||||
// LatencyFloor is the absolute change below which a latency move is
|
||||
// never status-affecting (default 20ms), so a proxy jittering around a
|
||||
// small latency doesn't write status forever.
|
||||
LatencyFloor time.Duration
|
||||
// ProbeTLSConfig overrides TLS verification for https probe URLs; nil
|
||||
// means system roots. Needed for private CAs (and tests).
|
||||
ProbeTLSConfig *tls.Config
|
||||
// Metrics, when non-nil, is fed on every probe — the status writes are
|
||||
// transition-only by design, so metrics are where high-frequency
|
||||
// signal (true probe recency, every latency sample) lives.
|
||||
Metrics ProbeMetrics
|
||||
|
||||
probeFn func(context.Context, *url.URL, crawlv1alpha1.HealthCheckSpec, *tls.Config) probeResult
|
||||
|
||||
mu sync.Mutex
|
||||
states map[types.NamespacedName]*state
|
||||
}
|
||||
|
||||
// NewEngine returns an Engine with a buffered Events channel, ready to be
|
||||
// handed to both mgr.Add and the reconciler (Health + HealthEvents fields).
|
||||
func NewEngine(reader client.Reader) *Engine {
|
||||
e := &Engine{Reader: reader}
|
||||
e.applyDefaults()
|
||||
return e
|
||||
}
|
||||
|
||||
func (e *Engine) applyDefaults() {
|
||||
if e.Events == nil {
|
||||
e.Events = make(chan event.GenericEvent, 64)
|
||||
}
|
||||
if e.Workers == 0 {
|
||||
e.Workers = 8
|
||||
}
|
||||
if e.Tick == 0 {
|
||||
e.Tick = time.Second
|
||||
}
|
||||
if e.MinReportInterval == 0 {
|
||||
e.MinReportInterval = time.Minute
|
||||
}
|
||||
if e.LatencyFloor == 0 {
|
||||
e.LatencyFloor = 20 * time.Millisecond
|
||||
}
|
||||
if e.probeFn == nil {
|
||||
e.probeFn = probe
|
||||
}
|
||||
if e.states == nil {
|
||||
e.states = map[types.NamespacedName]*state{}
|
||||
}
|
||||
}
|
||||
|
||||
// NeedLeaderElection makes the engine run only on the leader: probing from
|
||||
// every replica would multiply load on the proxies, and only the leader's
|
||||
// reconciler can represent the results anyway.
|
||||
func (e *Engine) NeedLeaderElection() bool { return true }
|
||||
|
||||
// Start runs the scheduler tick loop and the worker pool until ctx ends.
|
||||
func (e *Engine) Start(ctx context.Context) error {
|
||||
e.applyDefaults()
|
||||
jobs := make(chan probeJob)
|
||||
var wg sync.WaitGroup
|
||||
for range e.Workers {
|
||||
wg.Go(func() {
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case job := <-jobs:
|
||||
res := e.probeFn(ctx, job.proxyURL, job.hc, e.ProbeTLSConfig)
|
||||
e.record(job, res, time.Now())
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
ticker := time.NewTicker(e.Tick)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
wg.Wait()
|
||||
return nil
|
||||
case now := <-ticker.C:
|
||||
e.tick(ctx, now, jobs)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// tick lists proxies from the cache, refreshes the state map (create, seed,
|
||||
// prune, UID-mismatch reset), and hands due proxies to the worker pool.
|
||||
func (e *Engine) tick(ctx context.Context, now time.Time, jobs chan<- probeJob) {
|
||||
var list crawlv1alpha1.ProxyList
|
||||
if err := e.Reader.List(ctx, &list); err != nil {
|
||||
logf.FromContext(ctx).Error(err, "health engine: listing proxies")
|
||||
return
|
||||
}
|
||||
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
|
||||
probeable := make(map[types.NamespacedName]struct{}, len(list.Items))
|
||||
for i := range list.Items {
|
||||
p := &list.Items[i]
|
||||
host := p.EffectiveHost()
|
||||
if host == "" || !p.DeletionTimestamp.IsZero() {
|
||||
// Not probeable (provisioning, being replaced, or deleting).
|
||||
// Its state gets pruned below, so a replacement instance starts
|
||||
// with fresh counters.
|
||||
continue
|
||||
}
|
||||
key := client.ObjectKeyFromObject(p)
|
||||
probeable[key] = struct{}{}
|
||||
|
||||
hc := p.HealthCheckOrDefault()
|
||||
interval := time.Duration(hc.IntervalSeconds) * time.Second
|
||||
|
||||
st := e.states[key]
|
||||
if st == nil || st.uid != p.UID {
|
||||
// New proxy, or a delete+recreate under the same name — never
|
||||
// inherit the old object's counters.
|
||||
st = newState(p, now, interval)
|
||||
e.states[key] = st
|
||||
}
|
||||
if st.inFlight || now.Before(st.nextDue) {
|
||||
continue
|
||||
}
|
||||
job := probeJob{
|
||||
key: key,
|
||||
uid: p.UID,
|
||||
proxyURL: &url.URL{Scheme: "http", Host: net.JoinHostPort(host, strconv.Itoa(int(p.EffectivePort())))},
|
||||
hc: hc,
|
||||
interval: interval,
|
||||
}
|
||||
select {
|
||||
case jobs <- job:
|
||||
st.inFlight = true
|
||||
default:
|
||||
// Worker pool saturated; the proxy stays due and is retried on
|
||||
// the next tick.
|
||||
}
|
||||
}
|
||||
|
||||
for key := range e.states {
|
||||
if _, ok := probeable[key]; !ok {
|
||||
delete(e.states, key)
|
||||
if e.Metrics != nil {
|
||||
// Retire the per-proxy series with the state, or series
|
||||
// for deleted proxies leak forever.
|
||||
e.Metrics.ForgetProxy(key.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// newState seeds bookkeeping for a proxy the engine hasn't tracked yet. If
|
||||
// the CR already carries a Healthy verdict (leader handover, operator
|
||||
// restart), the verdict is kept — so a healthy proxy doesn't flap to
|
||||
// unknown — counters stay at zero so a real transition still needs a full
|
||||
// threshold run, and the first probe is jittered across the interval so a
|
||||
// restart doesn't fire the whole fleet's probes at once. A proxy with no
|
||||
// prior verdict is probed immediately.
|
||||
func newState(p *crawlv1alpha1.Proxy, now time.Time, interval time.Duration) *state {
|
||||
st := &state{uid: p.UID, nextDue: now}
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy)
|
||||
if cond == nil || cond.Status == metav1.ConditionUnknown {
|
||||
return st
|
||||
}
|
||||
healthy := cond.Status == metav1.ConditionTrue
|
||||
reported := healthy
|
||||
st.healthy = &healthy
|
||||
st.reportedHealthy = &reported
|
||||
st.latency = time.Duration(p.Status.LatencyMillis) * time.Millisecond
|
||||
st.reportedLatency = st.latency
|
||||
st.nextDue = now.Add(rand.N(interval))
|
||||
return st
|
||||
}
|
||||
|
||||
// record folds one probe result into the proxy's threshold state and emits
|
||||
// an event when the result is status-affecting: a first-ever verdict, a
|
||||
// threshold-crossing flip, or a material latency change (beyond
|
||||
// max(LatencyFloor, 50% of reported) and rate-limited by MinReportInterval).
|
||||
func (e *Engine) record(job probeJob, res probeResult, now time.Time) {
|
||||
if e.Metrics != nil {
|
||||
e.Metrics.ObserveProbe(job.key.String(), res.latency, res.ok)
|
||||
}
|
||||
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
|
||||
st := e.states[job.key]
|
||||
if st == nil || st.uid != job.uid {
|
||||
return // pruned or replaced while the probe was in flight
|
||||
}
|
||||
st.inFlight = false
|
||||
st.nextDue = now.Add(job.interval)
|
||||
st.lastProbe = now
|
||||
|
||||
if res.ok {
|
||||
st.consecOK++
|
||||
st.consecFail = 0
|
||||
st.latency = res.latency
|
||||
st.lastErr = ""
|
||||
} else {
|
||||
st.consecFail++
|
||||
st.consecOK = 0
|
||||
st.lastErr = res.err.Error()
|
||||
}
|
||||
|
||||
switch {
|
||||
case st.healthy == nil:
|
||||
healthy := res.ok
|
||||
st.healthy = &healthy
|
||||
case *st.healthy && st.consecFail >= job.hc.FailureThreshold:
|
||||
healthy := false
|
||||
st.healthy = &healthy
|
||||
case !*st.healthy && st.consecOK >= job.hc.SuccessThreshold:
|
||||
healthy := true
|
||||
st.healthy = &healthy
|
||||
}
|
||||
|
||||
var emit bool
|
||||
switch {
|
||||
case st.reportedHealthy == nil:
|
||||
emit = true
|
||||
case *st.reportedHealthy != *st.healthy:
|
||||
emit = true
|
||||
case res.ok && *st.healthy:
|
||||
// Latency-only updates matter only for a healthy verdict; a success
|
||||
// streak still below successThreshold must stay silent.
|
||||
delta := st.latency - st.reportedLatency
|
||||
if delta < 0 {
|
||||
delta = -delta
|
||||
}
|
||||
emit = delta > max(e.LatencyFloor, st.reportedLatency/2) &&
|
||||
now.Sub(st.lastReport) > e.MinReportInterval
|
||||
}
|
||||
if !emit {
|
||||
return
|
||||
}
|
||||
|
||||
evt := event.GenericEvent{Object: &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Namespace: job.key.Namespace, Name: job.key.Name},
|
||||
}}
|
||||
select {
|
||||
case e.Events <- evt:
|
||||
reported := *st.healthy
|
||||
st.reportedHealthy = &reported
|
||||
st.reportedLatency = st.latency
|
||||
st.lastReport = now
|
||||
default:
|
||||
// Channel full (reconciler wedged): drop, and deliberately do not
|
||||
// advance the reported markers, so the next probe retries the emit.
|
||||
}
|
||||
}
|
||||
|
||||
// Snapshot returns the engine's current verdict for key; ok is false while
|
||||
// no verdict exists (never probed, or state was reset).
|
||||
func (e *Engine) Snapshot(key types.NamespacedName) (Snapshot, bool) {
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
st := e.states[key]
|
||||
if st == nil || st.healthy == nil {
|
||||
return Snapshot{}, false
|
||||
}
|
||||
return Snapshot{
|
||||
Healthy: *st.healthy,
|
||||
Latency: st.latency,
|
||||
LastProbe: st.lastProbe,
|
||||
LastError: st.lastErr,
|
||||
ConsecutiveFailures: st.consecFail,
|
||||
}, true
|
||||
}
|
||||
372
internal/health/engine_test.go
Normal file
372
internal/health/engine_test.go
Normal file
@@ -0,0 +1,372 @@
|
||||
package health
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"errors"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
"sigs.k8s.io/controller-runtime/pkg/event"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
var testKey = types.NamespacedName{Namespace: "default", Name: "p1"}
|
||||
|
||||
func testEngine() *Engine {
|
||||
e := &Engine{Events: make(chan event.GenericEvent, 8)}
|
||||
e.applyDefaults()
|
||||
return e
|
||||
}
|
||||
|
||||
func testJob(failureThreshold, successThreshold int32) probeJob {
|
||||
return probeJob{
|
||||
key: testKey,
|
||||
uid: "uid-1",
|
||||
hc: crawlv1alpha1.HealthCheckSpec{
|
||||
FailureThreshold: failureThreshold,
|
||||
SuccessThreshold: successThreshold,
|
||||
},
|
||||
interval: 30 * time.Second,
|
||||
}
|
||||
}
|
||||
|
||||
func drainOneEvent(t *testing.T, e *Engine) event.GenericEvent {
|
||||
t.Helper()
|
||||
select {
|
||||
case evt := <-e.Events:
|
||||
return evt
|
||||
default:
|
||||
t.Fatal("expected an event, channel is empty")
|
||||
return event.GenericEvent{}
|
||||
}
|
||||
}
|
||||
|
||||
func assertNoEvent(t *testing.T, e *Engine) {
|
||||
t.Helper()
|
||||
select {
|
||||
case <-e.Events:
|
||||
t.Fatal("unexpected event emitted")
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
func boolPtr(b bool) *bool { return &b }
|
||||
|
||||
func TestRecord_firstResultEmits(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.states[testKey] = &state{uid: "uid-1"}
|
||||
|
||||
e.record(testJob(3, 1), probeResult{ok: true, latency: 30 * time.Millisecond}, time.Now())
|
||||
|
||||
evt := drainOneEvent(t, e)
|
||||
if got := evt.Object.GetName(); got != "p1" {
|
||||
t.Errorf("event object name = %q, want p1", got)
|
||||
}
|
||||
snap, ok := e.Snapshot(testKey)
|
||||
if !ok || !snap.Healthy {
|
||||
t.Errorf("Snapshot = %+v, %v; want healthy verdict", snap, ok)
|
||||
}
|
||||
if snap.Latency != 30*time.Millisecond {
|
||||
t.Errorf("latency = %v, want 30ms", snap.Latency)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecord_failureThresholdFlips(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.states[testKey] = &state{uid: "uid-1", healthy: boolPtr(true), reportedHealthy: boolPtr(true)}
|
||||
job := testJob(3, 1)
|
||||
probeErr := probeResult{err: errors.New("connect refused")}
|
||||
|
||||
e.record(job, probeErr, time.Now())
|
||||
e.record(job, probeErr, time.Now())
|
||||
assertNoEvent(t, e)
|
||||
if snap, _ := e.Snapshot(testKey); !snap.Healthy {
|
||||
t.Fatal("flipped unhealthy before failureThreshold was reached")
|
||||
}
|
||||
|
||||
e.record(job, probeErr, time.Now())
|
||||
drainOneEvent(t, e)
|
||||
snap, _ := e.Snapshot(testKey)
|
||||
if snap.Healthy {
|
||||
t.Error("still healthy after failureThreshold consecutive failures")
|
||||
}
|
||||
if snap.ConsecutiveFailures != 3 {
|
||||
t.Errorf("ConsecutiveFailures = %d, want 3", snap.ConsecutiveFailures)
|
||||
}
|
||||
if !strings.Contains(snap.LastError, "connect refused") {
|
||||
t.Errorf("LastError = %q, want the probe error", snap.LastError)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecord_successThresholdFlips(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.states[testKey] = &state{uid: "uid-1", healthy: boolPtr(false), reportedHealthy: boolPtr(false)}
|
||||
job := testJob(3, 2)
|
||||
success := probeResult{ok: true, latency: 25 * time.Millisecond}
|
||||
|
||||
e.record(job, success, time.Now())
|
||||
assertNoEvent(t, e)
|
||||
|
||||
e.record(job, success, time.Now())
|
||||
drainOneEvent(t, e)
|
||||
if snap, _ := e.Snapshot(testKey); !snap.Healthy {
|
||||
t.Error("not healthy after successThreshold consecutive successes")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecord_latencySuppression(t *testing.T) {
|
||||
t.Parallel()
|
||||
now := time.Now()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
reportedLatency time.Duration
|
||||
lastReport time.Time
|
||||
newLatency time.Duration
|
||||
wantEmit bool
|
||||
}{
|
||||
{
|
||||
name: "small change under the relative floor is suppressed",
|
||||
reportedLatency: 100 * time.Millisecond,
|
||||
lastReport: now.Add(-2 * time.Minute),
|
||||
newLatency: 110 * time.Millisecond,
|
||||
wantEmit: false,
|
||||
},
|
||||
{
|
||||
name: "small absolute jitter at low latency is suppressed",
|
||||
reportedLatency: 5 * time.Millisecond,
|
||||
lastReport: now.Add(-2 * time.Minute),
|
||||
newLatency: 20 * time.Millisecond, // >50% but under the 20ms floor
|
||||
wantEmit: false,
|
||||
},
|
||||
{
|
||||
name: "material change after the rate window emits",
|
||||
reportedLatency: 100 * time.Millisecond,
|
||||
lastReport: now.Add(-2 * time.Minute),
|
||||
newLatency: 200 * time.Millisecond,
|
||||
wantEmit: true,
|
||||
},
|
||||
{
|
||||
name: "material change inside the rate window is suppressed",
|
||||
reportedLatency: 100 * time.Millisecond,
|
||||
lastReport: now.Add(-10 * time.Second),
|
||||
newLatency: 400 * time.Millisecond,
|
||||
wantEmit: false,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.states[testKey] = &state{
|
||||
uid: "uid-1",
|
||||
healthy: boolPtr(true),
|
||||
reportedHealthy: boolPtr(true),
|
||||
reportedLatency: tc.reportedLatency,
|
||||
lastReport: tc.lastReport,
|
||||
}
|
||||
e.record(testJob(3, 1), probeResult{ok: true, latency: tc.newLatency}, now)
|
||||
if tc.wantEmit {
|
||||
drainOneEvent(t, e)
|
||||
} else {
|
||||
assertNoEvent(t, e)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecord_droppedEventIsRetried(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.Events = make(chan event.GenericEvent) // unbuffered, nobody reading
|
||||
e.states[testKey] = &state{uid: "uid-1"}
|
||||
success := probeResult{ok: true, latency: 30 * time.Millisecond}
|
||||
|
||||
e.record(testJob(3, 1), success, time.Now())
|
||||
e.mu.Lock()
|
||||
reported := e.states[testKey].reportedHealthy
|
||||
e.mu.Unlock()
|
||||
if reported != nil {
|
||||
t.Fatal("reported marker advanced although the event was dropped")
|
||||
}
|
||||
|
||||
// Channel drains (reconciler recovers): the next probe re-emits.
|
||||
e.Events = make(chan event.GenericEvent, 1)
|
||||
e.record(testJob(3, 1), success, time.Now())
|
||||
drainOneEvent(t, e)
|
||||
}
|
||||
|
||||
func TestRecord_staleJobIsIgnored(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.states[testKey] = &state{uid: "uid-NEW"}
|
||||
|
||||
job := testJob(3, 1)
|
||||
job.uid = "uid-OLD"
|
||||
e.record(job, probeResult{ok: true, latency: time.Millisecond}, time.Now())
|
||||
|
||||
assertNoEvent(t, e)
|
||||
if _, ok := e.Snapshot(testKey); ok {
|
||||
t.Error("stale probe produced a verdict for the new object")
|
||||
}
|
||||
}
|
||||
|
||||
func externalProxy(name, host string, mut ...func(*crawlv1alpha1.Proxy)) *crawlv1alpha1.Proxy {
|
||||
p := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: name, Namespace: "default", UID: types.UID("uid-" + name),
|
||||
},
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: host, Port: 3128},
|
||||
},
|
||||
}
|
||||
for _, m := range mut {
|
||||
m(p)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
func TestTick_schedulingAndPruning(t *testing.T) {
|
||||
t.Parallel()
|
||||
s := runtime.NewScheme()
|
||||
if err := crawlv1alpha1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("scheme: %v", err)
|
||||
}
|
||||
|
||||
probeable := externalProxy("probeable", "10.0.0.1")
|
||||
seeded := externalProxy("seeded", "10.0.0.2", func(p *crawlv1alpha1.Proxy) {
|
||||
p.Status.Conditions = []metav1.Condition{{
|
||||
Type: crawlv1alpha1.ConditionHealthy, Status: metav1.ConditionTrue,
|
||||
Reason: "ProbeSucceeded", LastTransitionTime: metav1.Now(),
|
||||
}}
|
||||
p.Status.LatencyMillis = 42
|
||||
})
|
||||
noIP := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "no-ip", Namespace: "default", UID: "uid-no-ip"},
|
||||
Spec: crawlv1alpha1.ProxySpec{Mode: crawlv1alpha1.ModeManaged, Provider: "stub"},
|
||||
}
|
||||
|
||||
e := testEngine()
|
||||
e.Reader = fake.NewClientBuilder().WithScheme(s).
|
||||
WithObjects(probeable, seeded, noIP).Build()
|
||||
// Stale entries: one for a proxy that no longer exists, one under a key
|
||||
// that now belongs to a different UID (delete + recreate).
|
||||
e.states[types.NamespacedName{Namespace: "default", Name: "gone"}] = &state{uid: "uid-gone"}
|
||||
e.states[types.NamespacedName{Namespace: "default", Name: "probeable"}] = &state{
|
||||
uid: "uid-previous-incarnation", healthy: boolPtr(false),
|
||||
}
|
||||
|
||||
jobs := make(chan probeJob, 8)
|
||||
e.tick(context.Background(), time.Now(), jobs)
|
||||
|
||||
var dispatched []probeJob
|
||||
for {
|
||||
select {
|
||||
case j := <-jobs:
|
||||
dispatched = append(dispatched, j)
|
||||
continue
|
||||
default:
|
||||
}
|
||||
break
|
||||
}
|
||||
|
||||
if len(dispatched) != 1 {
|
||||
t.Fatalf("dispatched %d jobs, want exactly 1 (only the fresh probeable proxy)", len(dispatched))
|
||||
}
|
||||
j := dispatched[0]
|
||||
if j.key.Name != "probeable" || j.uid != "uid-probeable" {
|
||||
t.Errorf("dispatched job = %+v, want the recreated probeable proxy", j)
|
||||
}
|
||||
if want := "http://" + "10.0.0.1:" + strconv.Itoa(3128); j.proxyURL.String() != want {
|
||||
t.Errorf("proxyURL = %s, want %s", j.proxyURL, want)
|
||||
}
|
||||
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
if _, ok := e.states[types.NamespacedName{Namespace: "default", Name: "gone"}]; ok {
|
||||
t.Error("state for a deleted proxy was not pruned")
|
||||
}
|
||||
if _, ok := e.states[types.NamespacedName{Namespace: "default", Name: "no-ip"}]; ok {
|
||||
t.Error("state was created for a proxy with no IP")
|
||||
}
|
||||
st := e.states[types.NamespacedName{Namespace: "default", Name: "probeable"}]
|
||||
if st == nil || st.uid != "uid-probeable" {
|
||||
t.Fatalf("state for recreated proxy = %+v, want fresh state with the new UID", st)
|
||||
}
|
||||
if st.healthy != nil && !*st.healthy {
|
||||
t.Error("recreated proxy inherited the previous incarnation's unhealthy verdict")
|
||||
}
|
||||
seededSt := e.states[types.NamespacedName{Namespace: "default", Name: "seeded"}]
|
||||
if seededSt == nil {
|
||||
t.Fatal("no state created for the seeded proxy")
|
||||
}
|
||||
if seededSt.healthy == nil || !*seededSt.healthy {
|
||||
t.Error("seeded proxy did not inherit its Healthy condition")
|
||||
}
|
||||
if seededSt.reportedHealthy == nil || !*seededSt.reportedHealthy {
|
||||
t.Error("seeded verdict must count as already reported, or restart would re-emit for the whole fleet")
|
||||
}
|
||||
if seededSt.reportedLatency != 42*time.Millisecond {
|
||||
t.Errorf("seeded reportedLatency = %v, want 42ms", seededSt.reportedLatency)
|
||||
}
|
||||
if seededSt.consecOK != 0 || seededSt.consecFail != 0 {
|
||||
t.Error("seeded counters must start at zero")
|
||||
}
|
||||
}
|
||||
|
||||
func TestEngine_StartEndToEnd(t *testing.T) {
|
||||
t.Parallel()
|
||||
s := runtime.NewScheme()
|
||||
if err := crawlv1alpha1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("scheme: %v", err)
|
||||
}
|
||||
|
||||
e := testEngine()
|
||||
e.Tick = 5 * time.Millisecond
|
||||
e.Reader = fake.NewClientBuilder().WithScheme(s).
|
||||
WithObjects(externalProxy("p1", "192.0.2.1")).Build()
|
||||
e.probeFn = func(context.Context, *url.URL, crawlv1alpha1.HealthCheckSpec, *tls.Config) probeResult {
|
||||
return probeResult{ok: true, latency: 12 * time.Millisecond}
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- e.Start(ctx) }()
|
||||
|
||||
select {
|
||||
case evt := <-e.Events:
|
||||
if evt.Object.GetName() != "p1" {
|
||||
t.Errorf("event for %q, want p1", evt.Object.GetName())
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("no health event within 5s")
|
||||
}
|
||||
snap, ok := e.Snapshot(types.NamespacedName{Namespace: "default", Name: "p1"})
|
||||
if !ok || !snap.Healthy || snap.Latency != 12*time.Millisecond {
|
||||
t.Errorf("Snapshot = %+v, %v; want healthy at 12ms", snap, ok)
|
||||
}
|
||||
|
||||
cancel()
|
||||
select {
|
||||
case err := <-done:
|
||||
if err != nil {
|
||||
t.Errorf("Start returned %v, want nil on context cancel", err)
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("Start did not stop within 5s of cancel")
|
||||
}
|
||||
}
|
||||
66
internal/health/probe.go
Normal file
66
internal/health/probe.go
Normal file
@@ -0,0 +1,66 @@
|
||||
// Package health actively probes every proxy by fetching a URL through the
|
||||
// proxy itself, keeps per-proxy threshold state, and pushes status-affecting
|
||||
// transitions to the reconciler over a channel. The engine owns health
|
||||
// state; the reconciler owns its representation in the Proxy's status.
|
||||
package health
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"fmt"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"slices"
|
||||
"time"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
// probeResult is the outcome of a single through-the-proxy probe.
|
||||
type probeResult struct {
|
||||
ok bool
|
||||
latency time.Duration
|
||||
err error
|
||||
}
|
||||
|
||||
// probe fetches hc.ProbeURL through the proxy at proxyURL. For an https
|
||||
// probe URL the transport issues CONNECT to the proxy and TLS-handshakes
|
||||
// through the tunnel; a proxy that accepts TCP but cannot egress answers
|
||||
// CONNECT with a non-200, which client.Do surfaces as an error, not a
|
||||
// response — so success requires err == nil AND an expected status code.
|
||||
// tlsCfg is nil in production (system roots); tests and private-CA setups
|
||||
// inject their own.
|
||||
func probe(ctx context.Context, proxyURL *url.URL, hc crawlv1alpha1.HealthCheckSpec, tlsCfg *tls.Config) probeResult {
|
||||
timeout := time.Duration(hc.TimeoutSeconds) * time.Second
|
||||
transport := &http.Transport{
|
||||
Proxy: http.ProxyURL(proxyURL),
|
||||
// Load-bearing: with keep-alives on, net/http caches the established
|
||||
// CONNECT tunnel and later probes would never re-exercise CONNECT —
|
||||
// exactly the failure this probe exists to catch.
|
||||
DisableKeepAlives: true,
|
||||
ForceAttemptHTTP2: false,
|
||||
TLSHandshakeTimeout: timeout,
|
||||
ResponseHeaderTimeout: timeout,
|
||||
TLSClientConfig: tlsCfg,
|
||||
DialContext: (&net.Dialer{Timeout: timeout}).DialContext,
|
||||
}
|
||||
defer transport.CloseIdleConnections()
|
||||
|
||||
client := &http.Client{Transport: transport, Timeout: timeout}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, hc.ProbeURL, nil)
|
||||
if err != nil {
|
||||
return probeResult{err: fmt.Errorf("building probe request: %w", err)}
|
||||
}
|
||||
start := time.Now()
|
||||
resp, err := client.Do(req)
|
||||
latency := time.Since(start)
|
||||
if err != nil {
|
||||
return probeResult{latency: latency, err: err}
|
||||
}
|
||||
defer func() { _ = resp.Body.Close() }()
|
||||
if !slices.Contains(hc.ExpectedStatusCodes, int32(resp.StatusCode)) {
|
||||
return probeResult{latency: latency, err: fmt.Errorf("unexpected status %d", resp.StatusCode)}
|
||||
}
|
||||
return probeResult{ok: true, latency: latency}
|
||||
}
|
||||
169
internal/health/probe_test.go
Normal file
169
internal/health/probe_test.go
Normal file
@@ -0,0 +1,169 @@
|
||||
package health
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"crypto/x509"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/url"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
// startConnectProxy runs a minimal but real HTTP proxy: CONNECT tunneling
|
||||
// for https targets, absolute-URI forwarding for plain http ones. With
|
||||
// refuseConnect it answers CONNECT with 502 — the "accepts TCP but cannot
|
||||
// egress" failure mode the probe must classify as unhealthy.
|
||||
func startConnectProxy(t *testing.T, refuseConnect bool) *httptest.Server {
|
||||
t.Helper()
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method == http.MethodConnect {
|
||||
if refuseConnect {
|
||||
http.Error(w, "no egress", http.StatusBadGateway)
|
||||
return
|
||||
}
|
||||
dst, err := net.DialTimeout("tcp", r.Host, time.Second)
|
||||
if err != nil {
|
||||
http.Error(w, err.Error(), http.StatusBadGateway)
|
||||
return
|
||||
}
|
||||
conn, bufrw, err := http.NewResponseController(w).Hijack()
|
||||
if err != nil {
|
||||
_ = dst.Close()
|
||||
t.Errorf("hijack: %v", err)
|
||||
return
|
||||
}
|
||||
_, _ = bufrw.WriteString("HTTP/1.1 200 Connection established\r\n\r\n")
|
||||
_ = bufrw.Flush()
|
||||
done := make(chan struct{}, 2)
|
||||
go func() { _, _ = io.Copy(dst, bufrw); done <- struct{}{} }()
|
||||
go func() { _, _ = io.Copy(conn, dst); done <- struct{}{} }()
|
||||
<-done
|
||||
_ = conn.Close()
|
||||
_ = dst.Close()
|
||||
return
|
||||
}
|
||||
out := r.Clone(r.Context())
|
||||
out.RequestURI = ""
|
||||
resp, err := http.DefaultTransport.RoundTrip(out)
|
||||
if err != nil {
|
||||
http.Error(w, err.Error(), http.StatusBadGateway)
|
||||
return
|
||||
}
|
||||
defer func() { _ = resp.Body.Close() }()
|
||||
for k, vv := range resp.Header {
|
||||
for _, v := range vv {
|
||||
w.Header().Add(k, v)
|
||||
}
|
||||
}
|
||||
w.WriteHeader(resp.StatusCode)
|
||||
_, _ = io.Copy(w, resp.Body)
|
||||
}))
|
||||
t.Cleanup(srv.Close)
|
||||
return srv
|
||||
}
|
||||
|
||||
func startTLSTarget(t *testing.T, status int) (*httptest.Server, *tls.Config) {
|
||||
t.Helper()
|
||||
target := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.WriteHeader(status)
|
||||
}))
|
||||
t.Cleanup(target.Close)
|
||||
pool := x509.NewCertPool()
|
||||
pool.AddCert(target.Certificate())
|
||||
return target, &tls.Config{RootCAs: pool}
|
||||
}
|
||||
|
||||
func proxyURL(t *testing.T, srv *httptest.Server) *url.URL {
|
||||
t.Helper()
|
||||
u, err := url.Parse(srv.URL)
|
||||
if err != nil {
|
||||
t.Fatalf("parsing proxy URL: %v", err)
|
||||
}
|
||||
return u
|
||||
}
|
||||
|
||||
func testHC(probeTarget string) crawlv1alpha1.HealthCheckSpec {
|
||||
return crawlv1alpha1.HealthCheckSpec{
|
||||
ProbeURL: probeTarget,
|
||||
IntervalSeconds: 30,
|
||||
TimeoutSeconds: 5,
|
||||
FailureThreshold: 3,
|
||||
SuccessThreshold: 1,
|
||||
ExpectedStatusCodes: []int32{200, 204},
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbe_connectTunnelSucceeds(t *testing.T) {
|
||||
t.Parallel()
|
||||
target, tlsCfg := startTLSTarget(t, http.StatusNoContent)
|
||||
proxy := startConnectProxy(t, false)
|
||||
|
||||
res := probe(context.Background(), proxyURL(t, proxy), testHC(target.URL), tlsCfg)
|
||||
if !res.ok {
|
||||
t.Fatalf("probe failed through working proxy: %v", res.err)
|
||||
}
|
||||
if res.latency <= 0 {
|
||||
t.Errorf("latency = %v, want > 0", res.latency)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbe_refusedConnectFails(t *testing.T) {
|
||||
t.Parallel()
|
||||
target, tlsCfg := startTLSTarget(t, http.StatusNoContent)
|
||||
proxy := startConnectProxy(t, true)
|
||||
|
||||
res := probe(context.Background(), proxyURL(t, proxy), testHC(target.URL), tlsCfg)
|
||||
if res.ok {
|
||||
t.Fatal("probe succeeded through a proxy that refuses CONNECT")
|
||||
}
|
||||
if res.err == nil {
|
||||
t.Error("expected an error from the refused CONNECT")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbe_unexpectedStatusFails(t *testing.T) {
|
||||
t.Parallel()
|
||||
target, tlsCfg := startTLSTarget(t, http.StatusInternalServerError)
|
||||
proxy := startConnectProxy(t, false)
|
||||
|
||||
res := probe(context.Background(), proxyURL(t, proxy), testHC(target.URL), tlsCfg)
|
||||
if res.ok {
|
||||
t.Fatal("probe succeeded on a 500 response")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbe_unreachableProxyFails(t *testing.T) {
|
||||
t.Parallel()
|
||||
// A listener that is immediately closed: guaranteed-refused port.
|
||||
l, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("reserving port: %v", err)
|
||||
}
|
||||
dead := &url.URL{Scheme: "http", Host: l.Addr().String()}
|
||||
_ = l.Close()
|
||||
|
||||
res := probe(context.Background(), dead, testHC("https://example.invalid/"), nil)
|
||||
if res.ok {
|
||||
t.Fatal("probe succeeded against a dead proxy")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbe_plainHTTPForwardSucceeds(t *testing.T) {
|
||||
t.Parallel()
|
||||
target := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}))
|
||||
t.Cleanup(target.Close)
|
||||
proxy := startConnectProxy(t, false)
|
||||
|
||||
res := probe(context.Background(), proxyURL(t, proxy), testHC(target.URL), nil)
|
||||
if !res.ok {
|
||||
t.Fatalf("plain-http probe failed: %v", res.err)
|
||||
}
|
||||
}
|
||||
319
internal/lease/store.go
Normal file
319
internal/lease/store.go
Normal file
@@ -0,0 +1,319 @@
|
||||
// Package lease implements the in-memory lease store behind the discovery
|
||||
// API: TTL-based proxy assignment with server-side usage tracking and
|
||||
// per-(proxy, target) cooldowns. Accepted prototype limitation, documented
|
||||
// in the README: state is per-process, so an operator restart drops all
|
||||
// leases and cooldowns — clients must tolerate a lease vanishing (their
|
||||
// requests still work; they just re-lease).
|
||||
package lease
|
||||
|
||||
import (
|
||||
"cmp"
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"errors"
|
||||
"fmt"
|
||||
"slices"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Result is a client's report of how a leased proxy behaved against a
|
||||
// target. ResultRateLimited and ResultBanned record a cooldown; ResultOK is
|
||||
// an acknowledgement and records nothing.
|
||||
type Result string
|
||||
|
||||
const (
|
||||
ResultOK Result = "ok"
|
||||
ResultRateLimited Result = "rate_limited"
|
||||
ResultBanned Result = "banned"
|
||||
)
|
||||
|
||||
// ParseResult maps a wire value to a Result; ok is false for anything
|
||||
// unknown, which the API layer turns into a 400.
|
||||
func ParseResult(s string) (Result, bool) {
|
||||
switch r := Result(s); r {
|
||||
case ResultOK, ResultRateLimited, ResultBanned:
|
||||
return r, true
|
||||
default:
|
||||
return "", false
|
||||
}
|
||||
}
|
||||
|
||||
var (
|
||||
// ErrNoMatch means no candidate could take a lease; AcquireStats says
|
||||
// why, and the API layer turns both into the 409 body.
|
||||
ErrNoMatch = errors.New("lease: no candidate available")
|
||||
// ErrUnknownLease means the lease ID does not resolve (404). Reports on
|
||||
// recently expired leases do NOT hit this — see the retention note on
|
||||
// Store.
|
||||
ErrUnknownLease = errors.New("lease: unknown lease id")
|
||||
)
|
||||
|
||||
// Candidate is one leasable proxy as seen by the caller at selection time.
|
||||
// The store itself knows nothing about Proxy objects — the discovery layer
|
||||
// filters for health/attributes and passes what selection needs.
|
||||
type Candidate struct {
|
||||
// Proxy is the opaque proxy key ("namespace/name").
|
||||
Proxy string
|
||||
// MaxLeases caps concurrent leases; 0 means unleasable.
|
||||
MaxLeases int32
|
||||
// Latency is the proxy's last reported latency, used as the tie-break.
|
||||
Latency time.Duration
|
||||
}
|
||||
|
||||
// Lease is a granted assignment. Values returned by the store are copies;
|
||||
// mutating them does not affect the store.
|
||||
type Lease struct {
|
||||
ID string
|
||||
Proxy string
|
||||
Target string
|
||||
ExpiresAt time.Time
|
||||
}
|
||||
|
||||
// AcquireRequest carries the candidate set and lease parameters. Acquire
|
||||
// deliberately takes the whole candidate set, not a pre-chosen proxy:
|
||||
// selection and insertion must happen under one lock, or two concurrent
|
||||
// requests both see "3 of 5 used" and overcommit.
|
||||
type AcquireRequest struct {
|
||||
Candidates []Candidate
|
||||
// Target scopes the cooldown check; empty means the global pool.
|
||||
Target string
|
||||
TTL time.Duration
|
||||
}
|
||||
|
||||
// AcquireStats explains an ErrNoMatch (and is returned on success too):
|
||||
// every candidate is either leased, at capacity, or in cooldown.
|
||||
type AcquireStats struct {
|
||||
Considered int
|
||||
AtCapacity int
|
||||
InCooldown int
|
||||
}
|
||||
|
||||
type cooldownKey struct{ proxy, target string }
|
||||
|
||||
// Store is the in-memory lease store. One mutex guards everything: at tens
|
||||
// of proxies and human-rate QPS, sharding would be premature complexity.
|
||||
//
|
||||
// Retention: an expired lease is kept for CooldownWindow past its TTL so a
|
||||
// Report arriving just after expiry still resolves — which matters most
|
||||
// exactly when a proxy is being rate-limited. Acquire and the counts ignore
|
||||
// retained leases; only the sweep finally drops them.
|
||||
type Store struct {
|
||||
// CooldownWindow is how long a reported proxy/target pair is excluded
|
||||
// from selection (default 15m; --lease-cooldown in Step 10).
|
||||
CooldownWindow time.Duration
|
||||
// SweepInterval is how often the expiry sweep runs (default 30s).
|
||||
SweepInterval time.Duration
|
||||
|
||||
now func() time.Time
|
||||
|
||||
mu sync.Mutex
|
||||
byID map[string]*Lease
|
||||
byProxy map[string]map[string]*Lease
|
||||
cooldowns map[cooldownKey]time.Time
|
||||
}
|
||||
|
||||
// NewStore returns a ready Store. A non-positive cooldownWindow selects the
|
||||
// 15-minute default.
|
||||
func NewStore(cooldownWindow time.Duration) *Store {
|
||||
if cooldownWindow <= 0 {
|
||||
cooldownWindow = 15 * time.Minute
|
||||
}
|
||||
return &Store{
|
||||
CooldownWindow: cooldownWindow,
|
||||
SweepInterval: 30 * time.Second,
|
||||
now: time.Now,
|
||||
byID: map[string]*Lease{},
|
||||
byProxy: map[string]map[string]*Lease{},
|
||||
cooldowns: map[cooldownKey]time.Time{},
|
||||
}
|
||||
}
|
||||
|
||||
// Acquire selects the least-loaded eligible candidate (ties: lowest
|
||||
// latency, then name, so selection is deterministic and testable) and
|
||||
// grants a lease on it.
|
||||
func (s *Store) Acquire(_ context.Context, req AcquireRequest) (*Lease, AcquireStats, error) {
|
||||
stats := AcquireStats{Considered: len(req.Candidates)}
|
||||
if req.TTL <= 0 {
|
||||
return nil, stats, fmt.Errorf("lease: non-positive TTL %v", req.TTL)
|
||||
}
|
||||
now := s.now()
|
||||
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
|
||||
type eligible struct {
|
||||
cand Candidate
|
||||
active int
|
||||
}
|
||||
var elig []eligible
|
||||
for _, c := range req.Candidates {
|
||||
if s.inCooldownLocked(c.Proxy, req.Target, now) {
|
||||
stats.InCooldown++
|
||||
continue
|
||||
}
|
||||
active := s.activeCountLocked(c.Proxy, now)
|
||||
if int32(active) >= c.MaxLeases {
|
||||
stats.AtCapacity++
|
||||
continue
|
||||
}
|
||||
elig = append(elig, eligible{cand: c, active: active})
|
||||
}
|
||||
if len(elig) == 0 {
|
||||
return nil, stats, ErrNoMatch
|
||||
}
|
||||
|
||||
slices.SortFunc(elig, func(a, b eligible) int {
|
||||
if c := cmp.Compare(a.active, b.active); c != 0 {
|
||||
return c
|
||||
}
|
||||
if c := cmp.Compare(a.cand.Latency, b.cand.Latency); c != 0 {
|
||||
return c
|
||||
}
|
||||
return strings.Compare(a.cand.Proxy, b.cand.Proxy)
|
||||
})
|
||||
|
||||
l := &Lease{
|
||||
ID: rand.Text(),
|
||||
Proxy: elig[0].cand.Proxy,
|
||||
Target: req.Target,
|
||||
ExpiresAt: now.Add(req.TTL),
|
||||
}
|
||||
s.byID[l.ID] = l
|
||||
if s.byProxy[l.Proxy] == nil {
|
||||
s.byProxy[l.Proxy] = map[string]*Lease{}
|
||||
}
|
||||
s.byProxy[l.Proxy][l.ID] = l
|
||||
|
||||
granted := *l
|
||||
return &granted, stats, nil
|
||||
}
|
||||
|
||||
// Release drops a lease early. Idempotent: releasing an unknown or already
|
||||
// expired lease is a no-op, so the API's DELETE can always answer 204.
|
||||
func (s *Store) Release(_ context.Context, id string) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
s.dropLocked(id)
|
||||
}
|
||||
|
||||
// Report records the outcome of using a lease. Rate-limited and banned
|
||||
// results put the (proxy, target) pair in cooldown — target taken from the
|
||||
// report, falling back to the lease's own target, falling back to the
|
||||
// global pool. Reports on recently expired leases still resolve (see the
|
||||
// retention note on Store).
|
||||
func (s *Store) Report(_ context.Context, id string, result Result, target string) error {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
l, ok := s.byID[id]
|
||||
if !ok {
|
||||
return ErrUnknownLease
|
||||
}
|
||||
if result == ResultOK {
|
||||
return nil
|
||||
}
|
||||
if target == "" {
|
||||
target = l.Target
|
||||
}
|
||||
s.cooldowns[cooldownKey{proxy: l.Proxy, target: target}] = s.now().Add(s.CooldownWindow)
|
||||
return nil
|
||||
}
|
||||
|
||||
// ActiveCount returns the number of unexpired leases held on one proxy.
|
||||
func (s *Store) ActiveCount(proxy string) int {
|
||||
now := s.now()
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
return s.activeCountLocked(proxy, now)
|
||||
}
|
||||
|
||||
// Counts returns the active-lease count per proxy, for the discovery list
|
||||
// endpoint and the metrics collector. Proxies with no active leases are
|
||||
// absent from the map.
|
||||
func (s *Store) Counts() map[string]int {
|
||||
now := s.now()
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
counts := make(map[string]int, len(s.byProxy))
|
||||
for proxy := range s.byProxy {
|
||||
if n := s.activeCountLocked(proxy, now); n > 0 {
|
||||
counts[proxy] = n
|
||||
}
|
||||
}
|
||||
return counts
|
||||
}
|
||||
|
||||
// Start runs the expiry sweep until ctx ends; it satisfies
|
||||
// manager.Runnable so cmd/main.go can mgr.Add the store directly.
|
||||
func (s *Store) Start(ctx context.Context) error {
|
||||
ticker := time.NewTicker(s.SweepInterval)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return nil
|
||||
case <-ticker.C:
|
||||
s.sweep(s.now())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// NeedLeaderElection is false: lease state is per-process and the discovery
|
||||
// API serves wherever this process runs, so the sweep must run there too.
|
||||
func (s *Store) NeedLeaderElection() bool { return false }
|
||||
|
||||
// sweep drops leases past their retention window and elapsed cooldowns.
|
||||
// Correctness never depends on sweep timing — every read path checks
|
||||
// expiry against the clock — so this is purely garbage collection.
|
||||
func (s *Store) sweep(now time.Time) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
for id, l := range s.byID {
|
||||
if now.After(l.ExpiresAt.Add(s.CooldownWindow)) {
|
||||
s.dropLocked(id)
|
||||
}
|
||||
}
|
||||
for k, until := range s.cooldowns {
|
||||
if now.After(until) {
|
||||
delete(s.cooldowns, k)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (s *Store) dropLocked(id string) {
|
||||
l, ok := s.byID[id]
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
delete(s.byID, id)
|
||||
delete(s.byProxy[l.Proxy], id)
|
||||
if len(s.byProxy[l.Proxy]) == 0 {
|
||||
delete(s.byProxy, l.Proxy)
|
||||
}
|
||||
}
|
||||
|
||||
func (s *Store) activeCountLocked(proxy string, now time.Time) int {
|
||||
n := 0
|
||||
for _, l := range s.byProxy[proxy] {
|
||||
if now.Before(l.ExpiresAt) {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// inCooldownLocked: the global cooldown (empty target) always applies; a
|
||||
// target-scoped cooldown additionally applies to acquisitions for that
|
||||
// target. An acquisition without a target sees only the global pool — a
|
||||
// proxy rate-limited by one site is still fine for everyone else.
|
||||
func (s *Store) inCooldownLocked(proxy, target string, now time.Time) bool {
|
||||
if until, ok := s.cooldowns[cooldownKey{proxy: proxy}]; ok && now.Before(until) {
|
||||
return true
|
||||
}
|
||||
if target == "" {
|
||||
return false
|
||||
}
|
||||
until, ok := s.cooldowns[cooldownKey{proxy: proxy, target: target}]
|
||||
return ok && now.Before(until)
|
||||
}
|
||||
365
internal/lease/store_test.go
Normal file
365
internal/lease/store_test.go
Normal file
@@ -0,0 +1,365 @@
|
||||
package lease
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// fakeClock is an injectable, manually advanced clock.
|
||||
type fakeClock struct {
|
||||
mu sync.Mutex
|
||||
cur time.Time
|
||||
}
|
||||
|
||||
func newFakeClock() *fakeClock {
|
||||
return &fakeClock{cur: time.Date(2026, 8, 9, 12, 0, 0, 0, time.UTC)}
|
||||
}
|
||||
|
||||
func (c *fakeClock) Now() time.Time {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
return c.cur
|
||||
}
|
||||
|
||||
func (c *fakeClock) Advance(d time.Duration) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
c.cur = c.cur.Add(d)
|
||||
}
|
||||
|
||||
func newTestStore() (*Store, *fakeClock) {
|
||||
s := NewStore(15 * time.Minute)
|
||||
clock := newFakeClock()
|
||||
s.now = clock.Now
|
||||
return s, clock
|
||||
}
|
||||
|
||||
func candidate(proxy string, maxLeases int32, latency time.Duration) Candidate {
|
||||
return Candidate{Proxy: proxy, MaxLeases: maxLeases, Latency: latency}
|
||||
}
|
||||
|
||||
func mustAcquire(t *testing.T, s *Store, req AcquireRequest) *Lease {
|
||||
t.Helper()
|
||||
l, _, err := s.Acquire(context.Background(), req)
|
||||
if err != nil {
|
||||
t.Fatalf("Acquire: %v", err)
|
||||
}
|
||||
return l
|
||||
}
|
||||
|
||||
func TestAcquire_capacity(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, _ := newTestStore()
|
||||
req := AcquireRequest{Candidates: []Candidate{candidate("ns/p1", 2, 0)}, TTL: time.Minute}
|
||||
|
||||
l1 := mustAcquire(t, s, req)
|
||||
l2 := mustAcquire(t, s, req)
|
||||
if l1.ID == l2.ID {
|
||||
t.Fatal("two leases share an ID")
|
||||
}
|
||||
if got := s.ActiveCount("ns/p1"); got != 2 {
|
||||
t.Fatalf("ActiveCount = %d, want 2", got)
|
||||
}
|
||||
|
||||
_, stats, err := s.Acquire(context.Background(), req)
|
||||
if !errors.Is(err, ErrNoMatch) {
|
||||
t.Fatalf("third acquire error = %v, want ErrNoMatch", err)
|
||||
}
|
||||
want := AcquireStats{Considered: 1, AtCapacity: 1}
|
||||
if stats != want {
|
||||
t.Errorf("stats = %+v, want %+v", stats, want)
|
||||
}
|
||||
|
||||
// Early release frees the slot again.
|
||||
s.Release(context.Background(), l1.ID)
|
||||
mustAcquire(t, s, req)
|
||||
}
|
||||
|
||||
func TestAcquire_maxLeasesZeroIsUnleasable(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, _ := newTestStore()
|
||||
_, stats, err := s.Acquire(context.Background(), AcquireRequest{
|
||||
Candidates: []Candidate{candidate("ns/p1", 0, 0)},
|
||||
TTL: time.Minute,
|
||||
})
|
||||
if !errors.Is(err, ErrNoMatch) {
|
||||
t.Fatalf("err = %v, want ErrNoMatch", err)
|
||||
}
|
||||
if stats.AtCapacity != 1 {
|
||||
t.Errorf("stats = %+v, want the unleasable proxy counted AtCapacity", stats)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAcquire_selectionOrder(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
t.Run("least loaded wins", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, _ := newTestStore()
|
||||
mustAcquire(t, s, AcquireRequest{
|
||||
Candidates: []Candidate{candidate("ns/a", 5, 10*time.Millisecond)}, TTL: time.Minute,
|
||||
})
|
||||
l := mustAcquire(t, s, AcquireRequest{
|
||||
Candidates: []Candidate{
|
||||
candidate("ns/a", 5, 10*time.Millisecond), // 1 active, lower latency
|
||||
candidate("ns/b", 5, 90*time.Millisecond), // 0 active
|
||||
},
|
||||
TTL: time.Minute,
|
||||
})
|
||||
if l.Proxy != "ns/b" {
|
||||
t.Errorf("chose %s, want the least-loaded ns/b", l.Proxy)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("latency breaks the load tie", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, _ := newTestStore()
|
||||
l := mustAcquire(t, s, AcquireRequest{
|
||||
Candidates: []Candidate{
|
||||
candidate("ns/a", 5, 90*time.Millisecond),
|
||||
candidate("ns/b", 5, 10*time.Millisecond),
|
||||
},
|
||||
TTL: time.Minute,
|
||||
})
|
||||
if l.Proxy != "ns/b" {
|
||||
t.Errorf("chose %s, want the lower-latency ns/b", l.Proxy)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("name breaks a full tie deterministically", func(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, _ := newTestStore()
|
||||
l := mustAcquire(t, s, AcquireRequest{
|
||||
Candidates: []Candidate{
|
||||
candidate("ns/b", 5, 10*time.Millisecond),
|
||||
candidate("ns/a", 5, 10*time.Millisecond),
|
||||
},
|
||||
TTL: time.Minute,
|
||||
})
|
||||
if l.Proxy != "ns/a" {
|
||||
t.Errorf("chose %s, want ns/a (lexicographic tie-break)", l.Proxy)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestAcquire_cooldownScoping(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, _ := newTestStore()
|
||||
cands := []Candidate{candidate("ns/p1", 5, 0)}
|
||||
|
||||
l := mustAcquire(t, s, AcquireRequest{Candidates: cands, Target: "example.com", TTL: time.Minute})
|
||||
if err := s.Report(context.Background(), l.ID, ResultRateLimited, "example.com"); err != nil {
|
||||
t.Fatalf("Report: %v", err)
|
||||
}
|
||||
|
||||
// Same target: excluded.
|
||||
_, stats, err := s.Acquire(context.Background(), AcquireRequest{
|
||||
Candidates: cands, Target: "example.com", TTL: time.Minute,
|
||||
})
|
||||
if !errors.Is(err, ErrNoMatch) || stats.InCooldown != 1 {
|
||||
t.Errorf("same-target acquire = (%v, %+v), want ErrNoMatch with InCooldown=1", err, stats)
|
||||
}
|
||||
|
||||
// Different target: fine.
|
||||
mustAcquire(t, s, AcquireRequest{Candidates: cands, Target: "other.org", TTL: time.Minute})
|
||||
|
||||
// No target (global pool): a target-scoped cooldown does not apply.
|
||||
mustAcquire(t, s, AcquireRequest{Candidates: cands, TTL: time.Minute})
|
||||
}
|
||||
|
||||
func TestAcquire_globalCooldownBlocksEverything(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, _ := newTestStore()
|
||||
cands := []Candidate{candidate("ns/p1", 5, 0)}
|
||||
|
||||
// A lease without a target, reported banned without a target: the
|
||||
// cooldown lands on the global pool.
|
||||
l := mustAcquire(t, s, AcquireRequest{Candidates: cands, TTL: time.Minute})
|
||||
if err := s.Report(context.Background(), l.ID, ResultBanned, ""); err != nil {
|
||||
t.Fatalf("Report: %v", err)
|
||||
}
|
||||
|
||||
for _, target := range []string{"", "example.com"} {
|
||||
_, stats, err := s.Acquire(context.Background(), AcquireRequest{
|
||||
Candidates: cands, Target: target, TTL: time.Minute,
|
||||
})
|
||||
if !errors.Is(err, ErrNoMatch) || stats.InCooldown != 1 {
|
||||
t.Errorf("acquire(target=%q) = (%v, %+v), want global cooldown to block", target, err, stats)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAcquire_cooldownExpires(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, clock := newTestStore()
|
||||
cands := []Candidate{candidate("ns/p1", 5, 0)}
|
||||
|
||||
l := mustAcquire(t, s, AcquireRequest{Candidates: cands, TTL: time.Minute})
|
||||
if err := s.Report(context.Background(), l.ID, ResultRateLimited, ""); err != nil {
|
||||
t.Fatalf("Report: %v", err)
|
||||
}
|
||||
if _, _, err := s.Acquire(context.Background(), AcquireRequest{Candidates: cands, TTL: time.Minute}); !errors.Is(err, ErrNoMatch) {
|
||||
t.Fatal("expected cooldown to block immediately after the report")
|
||||
}
|
||||
|
||||
clock.Advance(15*time.Minute + time.Second)
|
||||
mustAcquire(t, s, AcquireRequest{Candidates: cands, TTL: time.Minute})
|
||||
}
|
||||
|
||||
func TestExpiry_freesCapacityWithoutSweep(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, clock := newTestStore()
|
||||
req := AcquireRequest{Candidates: []Candidate{candidate("ns/p1", 1, 0)}, TTL: time.Minute}
|
||||
|
||||
mustAcquire(t, s, req)
|
||||
if _, _, err := s.Acquire(context.Background(), req); !errors.Is(err, ErrNoMatch) {
|
||||
t.Fatal("capacity 1 not enforced")
|
||||
}
|
||||
|
||||
clock.Advance(2 * time.Minute)
|
||||
// No sweep has run; expiry must still free capacity and zero the counts.
|
||||
if got := s.ActiveCount("ns/p1"); got != 0 {
|
||||
t.Fatalf("ActiveCount after TTL = %d, want 0", got)
|
||||
}
|
||||
if counts := s.Counts(); len(counts) != 0 {
|
||||
t.Fatalf("Counts after TTL = %v, want empty", counts)
|
||||
}
|
||||
mustAcquire(t, s, req)
|
||||
}
|
||||
|
||||
func TestReport_expiredButRetainedLease(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, clock := newTestStore()
|
||||
cands := []Candidate{candidate("ns/p1", 5, 0)}
|
||||
|
||||
l := mustAcquire(t, s, AcquireRequest{Candidates: cands, Target: "example.com", TTL: time.Minute})
|
||||
|
||||
// TTL lapses; the report arrives late — exactly when the proxy is being
|
||||
// rate-limited, which is when the cooldown matters most.
|
||||
clock.Advance(5 * time.Minute)
|
||||
s.sweep(clock.Now())
|
||||
if err := s.Report(context.Background(), l.ID, ResultRateLimited, ""); err != nil {
|
||||
t.Fatalf("Report on an expired-but-retained lease: %v", err)
|
||||
}
|
||||
// The cooldown fell back to the lease's own target.
|
||||
_, stats, err := s.Acquire(context.Background(), AcquireRequest{
|
||||
Candidates: cands, Target: "example.com", TTL: time.Minute,
|
||||
})
|
||||
if !errors.Is(err, ErrNoMatch) || stats.InCooldown != 1 {
|
||||
t.Errorf("acquire = (%v, %+v), want cooldown from the late report", err, stats)
|
||||
}
|
||||
|
||||
// Past the retention window the sweep finally drops it.
|
||||
clock.Advance(15 * time.Minute)
|
||||
s.sweep(clock.Now())
|
||||
if err := s.Report(context.Background(), l.ID, ResultRateLimited, ""); !errors.Is(err, ErrUnknownLease) {
|
||||
t.Errorf("Report after retention = %v, want ErrUnknownLease", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReport_okRecordsNothing(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, _ := newTestStore()
|
||||
cands := []Candidate{candidate("ns/p1", 5, 0)}
|
||||
l := mustAcquire(t, s, AcquireRequest{Candidates: cands, Target: "example.com", TTL: time.Minute})
|
||||
|
||||
if err := s.Report(context.Background(), l.ID, ResultOK, "example.com"); err != nil {
|
||||
t.Fatalf("Report(ok): %v", err)
|
||||
}
|
||||
mustAcquire(t, s, AcquireRequest{Candidates: cands, Target: "example.com", TTL: time.Minute})
|
||||
}
|
||||
|
||||
func TestRelease_isIdempotent(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, _ := newTestStore()
|
||||
l := mustAcquire(t, s, AcquireRequest{Candidates: []Candidate{candidate("ns/p1", 1, 0)}, TTL: time.Minute})
|
||||
|
||||
s.Release(context.Background(), l.ID)
|
||||
s.Release(context.Background(), l.ID)
|
||||
s.Release(context.Background(), "never-existed")
|
||||
if got := s.ActiveCount("ns/p1"); got != 0 {
|
||||
t.Errorf("ActiveCount = %d, want 0", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseResult(t *testing.T) {
|
||||
t.Parallel()
|
||||
for _, valid := range []string{"ok", "rate_limited", "banned"} {
|
||||
if _, ok := ParseResult(valid); !ok {
|
||||
t.Errorf("ParseResult(%q) rejected a valid value", valid)
|
||||
}
|
||||
}
|
||||
for _, invalid := range []string{"", "OK", "throttled", "rate-limited"} {
|
||||
if _, ok := ParseResult(invalid); ok {
|
||||
t.Errorf("ParseResult(%q) accepted an invalid value", invalid)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAcquire_concurrentNeverOvercommits(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, _ := newTestStore()
|
||||
req := AcquireRequest{Candidates: []Candidate{candidate("ns/p1", 5, 0)}, TTL: time.Minute}
|
||||
|
||||
const attempts = 40
|
||||
var wg sync.WaitGroup
|
||||
granted := make(chan *Lease, attempts)
|
||||
for range attempts {
|
||||
wg.Go(func() {
|
||||
if l, _, err := s.Acquire(context.Background(), req); err == nil {
|
||||
granted <- l
|
||||
}
|
||||
})
|
||||
}
|
||||
wg.Wait()
|
||||
close(granted)
|
||||
|
||||
var n int
|
||||
for range granted {
|
||||
n++
|
||||
}
|
||||
if n != 5 {
|
||||
t.Errorf("%d of %d concurrent acquires granted, want exactly MaxLeases=5", n, attempts)
|
||||
}
|
||||
if got := s.ActiveCount("ns/p1"); got != 5 {
|
||||
t.Errorf("ActiveCount = %d, want 5", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStart_sweepsAndStops(t *testing.T) {
|
||||
t.Parallel()
|
||||
s, clock := newTestStore()
|
||||
s.SweepInterval = time.Millisecond
|
||||
|
||||
l := mustAcquire(t, s, AcquireRequest{Candidates: []Candidate{candidate("ns/p1", 5, 0)}, TTL: time.Minute})
|
||||
clock.Advance(20 * time.Minute) // past TTL + retention
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- s.Start(ctx) }()
|
||||
|
||||
deadline := time.After(5 * time.Second)
|
||||
for {
|
||||
if err := s.Report(context.Background(), l.ID, ResultOK, ""); errors.Is(err, ErrUnknownLease) {
|
||||
break
|
||||
}
|
||||
select {
|
||||
case <-deadline:
|
||||
t.Fatal("sweep never dropped the lease")
|
||||
case <-time.After(5 * time.Millisecond):
|
||||
}
|
||||
}
|
||||
|
||||
cancel()
|
||||
select {
|
||||
case err := <-done:
|
||||
if err != nil {
|
||||
t.Errorf("Start returned %v, want nil", err)
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("Start did not stop on cancel")
|
||||
}
|
||||
}
|
||||
119
internal/metrics/metrics.go
Normal file
119
internal/metrics/metrics.go
Normal file
@@ -0,0 +1,119 @@
|
||||
// Package metrics defines the operator's Prometheus metrics. Nothing here
|
||||
// registers itself — no init(), per house rules — the composition root
|
||||
// calls Register explicitly, which also lets every test use a fresh
|
||||
// registry.
|
||||
package metrics
|
||||
|
||||
import (
|
||||
"time"
|
||||
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
)
|
||||
|
||||
// Metrics holds the vector metrics the operator's components feed. The
|
||||
// consuming packages (health, discovery, provider) each define their own
|
||||
// small recorder interface; *Metrics satisfies all of them structurally,
|
||||
// so none of them import this package's prometheus surface.
|
||||
type Metrics struct {
|
||||
healthcheckDuration *prometheus.HistogramVec
|
||||
healthcheckFailures *prometheus.CounterVec
|
||||
leaseRequests *prometheus.CounterVec
|
||||
providerRequests *prometheus.CounterVec
|
||||
}
|
||||
|
||||
// New builds the metric set, unregistered.
|
||||
func New() *Metrics {
|
||||
return &Metrics{
|
||||
healthcheckDuration: prometheus.NewHistogramVec(prometheus.HistogramOpts{
|
||||
Name: "proxy_operator_healthcheck_duration_seconds",
|
||||
Help: "Duration of through-the-proxy health probes.",
|
||||
Buckets: prometheus.DefBuckets,
|
||||
}, []string{"proxy"}),
|
||||
healthcheckFailures: prometheus.NewCounterVec(prometheus.CounterOpts{
|
||||
Name: "proxy_operator_healthcheck_failures_total",
|
||||
Help: "Failed health probes.",
|
||||
}, []string{"proxy"}),
|
||||
leaseRequests: prometheus.NewCounterVec(prometheus.CounterOpts{
|
||||
Name: "proxy_operator_lease_requests_total",
|
||||
Help: "Lease acquisition requests by outcome.",
|
||||
}, []string{"outcome"}),
|
||||
providerRequests: prometheus.NewCounterVec(prometheus.CounterOpts{
|
||||
Name: "proxy_operator_provider_requests_total",
|
||||
Help: "Provider API calls by operation and classified result.",
|
||||
}, []string{"provider", "op", "result"}),
|
||||
}
|
||||
}
|
||||
|
||||
// Register registers the vectors plus the two scrape-time collectors.
|
||||
// proxyPhases and activeLeases are read at every scrape: gauges derived
|
||||
// from reconcile-time increments inevitably drift and leak series on
|
||||
// delete; reading the source of truth cannot.
|
||||
func (m *Metrics) Register(reg prometheus.Registerer, proxyPhases func() map[string]int, activeLeases func() int) error {
|
||||
collectors := []prometheus.Collector{
|
||||
m.healthcheckDuration,
|
||||
m.healthcheckFailures,
|
||||
m.leaseRequests,
|
||||
m.providerRequests,
|
||||
&constCollector{
|
||||
desc: prometheus.NewDesc("proxy_operator_proxies",
|
||||
"Proxy objects by phase.", []string{"phase"}, nil),
|
||||
read: proxyPhases,
|
||||
},
|
||||
&constCollector{
|
||||
desc: prometheus.NewDesc("proxy_operator_leases_active",
|
||||
"Currently active leases.", nil, nil),
|
||||
read: func() map[string]int { return map[string]int{"": activeLeases()} },
|
||||
},
|
||||
}
|
||||
for _, c := range collectors {
|
||||
if err := reg.Register(c); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ObserveProbe records one health probe. Called on every probe — metrics
|
||||
// are the home for high-frequency signal that must never touch status.
|
||||
func (m *Metrics) ObserveProbe(proxy string, latency time.Duration, success bool) {
|
||||
m.healthcheckDuration.WithLabelValues(proxy).Observe(latency.Seconds())
|
||||
if !success {
|
||||
m.healthcheckFailures.WithLabelValues(proxy).Inc()
|
||||
}
|
||||
}
|
||||
|
||||
// ForgetProxy drops the per-proxy series when the health engine prunes its
|
||||
// state — without this, series for deleted proxies leak forever.
|
||||
func (m *Metrics) ForgetProxy(proxy string) {
|
||||
m.healthcheckDuration.DeleteLabelValues(proxy)
|
||||
m.healthcheckFailures.DeleteLabelValues(proxy)
|
||||
}
|
||||
|
||||
// LeaseRequest records a lease acquisition outcome ("granted"|"no_match").
|
||||
func (m *Metrics) LeaseRequest(outcome string) {
|
||||
m.leaseRequests.WithLabelValues(outcome).Inc()
|
||||
}
|
||||
|
||||
// ProviderRequest records one provider API call with its classified result.
|
||||
func (m *Metrics) ProviderRequest(provider, op, result string) {
|
||||
m.providerRequests.WithLabelValues(provider, op, result).Inc()
|
||||
}
|
||||
|
||||
// constCollector reads a label→value map at scrape time and emits one
|
||||
// gauge sample per entry. An empty-string label key means "no labels".
|
||||
type constCollector struct {
|
||||
desc *prometheus.Desc
|
||||
read func() map[string]int
|
||||
}
|
||||
|
||||
func (c *constCollector) Describe(ch chan<- *prometheus.Desc) { ch <- c.desc }
|
||||
|
||||
func (c *constCollector) Collect(ch chan<- prometheus.Metric) {
|
||||
for label, value := range c.read() {
|
||||
if label == "" {
|
||||
ch <- prometheus.MustNewConstMetric(c.desc, prometheus.GaugeValue, float64(value))
|
||||
continue
|
||||
}
|
||||
ch <- prometheus.MustNewConstMetric(c.desc, prometheus.GaugeValue, float64(value), label)
|
||||
}
|
||||
}
|
||||
113
internal/metrics/metrics_test.go
Normal file
113
internal/metrics/metrics_test.go
Normal file
@@ -0,0 +1,113 @@
|
||||
package metrics
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
"github.com/prometheus/client_golang/prometheus/testutil"
|
||||
)
|
||||
|
||||
// register wires a fresh registry — the reason Register exists instead of
|
||||
// init()-time self-registration.
|
||||
func register(t *testing.T, m *Metrics, phases map[string]int, active int) *prometheus.Registry {
|
||||
t.Helper()
|
||||
reg := prometheus.NewRegistry()
|
||||
err := m.Register(reg,
|
||||
func() map[string]int { return phases },
|
||||
func() int { return active },
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("Register: %v", err)
|
||||
}
|
||||
return reg
|
||||
}
|
||||
|
||||
func TestRegister_scrapeTimeCollectors(t *testing.T) {
|
||||
t.Parallel()
|
||||
m := New()
|
||||
reg := register(t, m, map[string]int{"Ready": 3, "Provisioning": 1}, 7)
|
||||
|
||||
expected := `
|
||||
# HELP proxy_operator_leases_active Currently active leases.
|
||||
# TYPE proxy_operator_leases_active gauge
|
||||
proxy_operator_leases_active 7
|
||||
# HELP proxy_operator_proxies Proxy objects by phase.
|
||||
# TYPE proxy_operator_proxies gauge
|
||||
proxy_operator_proxies{phase="Provisioning"} 1
|
||||
proxy_operator_proxies{phase="Ready"} 3
|
||||
`
|
||||
if err := testutil.GatherAndCompare(reg, strings.NewReader(expected),
|
||||
"proxy_operator_proxies", "proxy_operator_leases_active"); err != nil {
|
||||
t.Error(err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestObserveProbe_andForget(t *testing.T) {
|
||||
t.Parallel()
|
||||
m := New()
|
||||
reg := register(t, m, nil, 0)
|
||||
|
||||
m.ObserveProbe("default/p1", 30*time.Millisecond, true)
|
||||
m.ObserveProbe("default/p1", 40*time.Millisecond, false)
|
||||
m.ObserveProbe("default/p2", 10*time.Millisecond, true)
|
||||
|
||||
if got := testutil.CollectAndCount(m.healthcheckDuration); got != 2 {
|
||||
t.Errorf("duration series = %d, want 2 (one per proxy)", got)
|
||||
}
|
||||
if got := testutil.ToFloat64(m.healthcheckFailures.WithLabelValues("default/p1")); got != 1 {
|
||||
t.Errorf("p1 failures = %v, want 1 (only the failed probe)", got)
|
||||
}
|
||||
|
||||
m.ForgetProxy("default/p1")
|
||||
if got := testutil.CollectAndCount(m.healthcheckDuration); got != 1 {
|
||||
t.Errorf("duration series after ForgetProxy = %d, want 1 — series must not leak", got)
|
||||
}
|
||||
if got := testutil.CollectAndCount(m.healthcheckFailures); got != 0 {
|
||||
t.Errorf("failure series after ForgetProxy = %d, want 0", got)
|
||||
}
|
||||
_ = reg
|
||||
}
|
||||
|
||||
func TestLeaseRequest(t *testing.T) {
|
||||
t.Parallel()
|
||||
m := New()
|
||||
register(t, m, nil, 0)
|
||||
|
||||
m.LeaseRequest("granted")
|
||||
m.LeaseRequest("granted")
|
||||
m.LeaseRequest("no_match")
|
||||
|
||||
if got := testutil.ToFloat64(m.leaseRequests.WithLabelValues("granted")); got != 2 {
|
||||
t.Errorf("granted = %v, want 2", got)
|
||||
}
|
||||
if got := testutil.ToFloat64(m.leaseRequests.WithLabelValues("no_match")); got != 1 {
|
||||
t.Errorf("no_match = %v, want 1", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProviderRequest(t *testing.T) {
|
||||
t.Parallel()
|
||||
m := New()
|
||||
register(t, m, nil, 0)
|
||||
|
||||
m.ProviderRequest("gcp-eu", "create", "ok")
|
||||
m.ProviderRequest("gcp-eu", "create", "quota_exceeded")
|
||||
|
||||
if got := testutil.ToFloat64(m.providerRequests.WithLabelValues("gcp-eu", "create", "ok")); got != 1 {
|
||||
t.Errorf("ok = %v, want 1", got)
|
||||
}
|
||||
if got := testutil.ToFloat64(m.providerRequests.WithLabelValues("gcp-eu", "create", "quota_exceeded")); got != 1 {
|
||||
t.Errorf("quota_exceeded = %v, want 1", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegister_freshRegistryPerTest(t *testing.T) {
|
||||
t.Parallel()
|
||||
// Registering the same metric set on two registries must both succeed —
|
||||
// the property init()-style global registration would break.
|
||||
m1, m2 := New(), New()
|
||||
register(t, m1, nil, 0)
|
||||
register(t, m2, nil, 0)
|
||||
}
|
||||
103
internal/provider/config.go
Normal file
103
internal/provider/config.go
Normal file
@@ -0,0 +1,103 @@
|
||||
package provider
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
|
||||
"sigs.k8s.io/yaml"
|
||||
)
|
||||
|
||||
// Config is the top-level shape of the --providers-config file.
|
||||
type Config struct {
|
||||
Providers []ProviderConfig `json:"providers"`
|
||||
}
|
||||
|
||||
// ProviderConfig is one named, typed provider instance — e.g. "gcp-eu" and
|
||||
// "gcp-us" can be two ProviderConfigs of Type "gcp" with different GCP
|
||||
// blocks. Exactly one of the type-specific blocks below should be set,
|
||||
// matching Type.
|
||||
type ProviderConfig struct {
|
||||
Name string `json:"name"`
|
||||
Type string `json:"type"`
|
||||
Kubernetes *KubernetesConfig `json:"kubernetes,omitempty"`
|
||||
GCP *GCPConfig `json:"gcp,omitempty"`
|
||||
}
|
||||
|
||||
// KubernetesConfig configures the kubernetes-pod provider, which creates
|
||||
// proxy Pods in the same cluster the operator itself runs in.
|
||||
type KubernetesConfig struct {
|
||||
// Image is the proxy container image. Default "ubuntu/squid:6.6-24.04_edge".
|
||||
Image string `json:"image,omitempty"`
|
||||
}
|
||||
|
||||
// GCPConfig configures a named GCP provider instance.
|
||||
type GCPConfig struct {
|
||||
// Project is the GCP project ID. Required.
|
||||
Project string `json:"project"`
|
||||
// Network is the VPC network name. Default "default".
|
||||
Network string `json:"network,omitempty"`
|
||||
// NetworkTag is the firewall/network tag applied to created instances.
|
||||
// Default "proxy-operator".
|
||||
NetworkTag string `json:"networkTag,omitempty"`
|
||||
// DiskSizeGB is the boot disk size in GB. Default 10.
|
||||
DiskSizeGB int64 `json:"diskSizeGb,omitempty"`
|
||||
}
|
||||
|
||||
// LoadConfigFile reads and parses a providers-config file from disk.
|
||||
func LoadConfigFile(path string) (*Config, error) {
|
||||
data, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("reading providers config %s: %w", path, err)
|
||||
}
|
||||
return LoadConfig(data)
|
||||
}
|
||||
|
||||
// LoadConfig parses and validates providers-config YAML. It fails fast:
|
||||
// unknown provider types, duplicate names, and missing type-specific
|
||||
// required fields are all load-time errors here, not runtime surprises
|
||||
// discovered only when a provider is actually used.
|
||||
func LoadConfig(data []byte) (*Config, error) {
|
||||
var cfg Config
|
||||
if err := yaml.UnmarshalStrict(data, &cfg); err != nil {
|
||||
return nil, fmt.Errorf("parsing providers config: %w", err)
|
||||
}
|
||||
if err := cfg.validate(); err != nil {
|
||||
return nil, fmt.Errorf("validating providers config: %w", err)
|
||||
}
|
||||
return &cfg, nil
|
||||
}
|
||||
|
||||
func (c *Config) validate() error {
|
||||
if len(c.Providers) == 0 {
|
||||
return fmt.Errorf("providers: at least one provider must be configured")
|
||||
}
|
||||
seen := make(map[string]bool, len(c.Providers))
|
||||
for i, p := range c.Providers {
|
||||
if p.Name == "" {
|
||||
return fmt.Errorf("providers[%d]: name is required", i)
|
||||
}
|
||||
if seen[p.Name] {
|
||||
return fmt.Errorf("providers[%d]: duplicate provider name %q", i, p.Name)
|
||||
}
|
||||
seen[p.Name] = true
|
||||
|
||||
switch p.Type {
|
||||
case "kubernetes":
|
||||
if p.GCP != nil {
|
||||
return fmt.Errorf("providers[%d] %q: type is kubernetes but a gcp block is set", i, p.Name)
|
||||
}
|
||||
case "gcp":
|
||||
if p.Kubernetes != nil {
|
||||
return fmt.Errorf("providers[%d] %q: type is gcp but a kubernetes block is set", i, p.Name)
|
||||
}
|
||||
if p.GCP == nil || p.GCP.Project == "" {
|
||||
return fmt.Errorf("providers[%d] %q: gcp.project is required", i, p.Name)
|
||||
}
|
||||
case "":
|
||||
return fmt.Errorf("providers[%d] %q: type is required", i, p.Name)
|
||||
default:
|
||||
return fmt.Errorf("providers[%d] %q: unknown provider type %q", i, p.Name, p.Type)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
163
internal/provider/config_test.go
Normal file
163
internal/provider/config_test.go
Normal file
@@ -0,0 +1,163 @@
|
||||
package provider
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestLoadConfig_valid(t *testing.T) {
|
||||
t.Parallel()
|
||||
data := []byte(`
|
||||
providers:
|
||||
- name: kubernetes
|
||||
type: kubernetes
|
||||
kubernetes:
|
||||
image: ubuntu/squid:6.6-24.04_edge
|
||||
- name: gcp-eu
|
||||
type: gcp
|
||||
gcp:
|
||||
project: my-project
|
||||
network: custom-net
|
||||
`)
|
||||
cfg, err := LoadConfig(data)
|
||||
if err != nil {
|
||||
t.Fatalf("LoadConfig() error = %v, want nil", err)
|
||||
}
|
||||
if len(cfg.Providers) != 2 {
|
||||
t.Fatalf("len(cfg.Providers) = %d, want 2", len(cfg.Providers))
|
||||
}
|
||||
if cfg.Providers[0].Kubernetes == nil || cfg.Providers[0].Kubernetes.Image != "ubuntu/squid:6.6-24.04_edge" {
|
||||
t.Errorf("providers[0].kubernetes = %+v, want Image=ubuntu/squid:6.6-24.04_edge", cfg.Providers[0].Kubernetes)
|
||||
}
|
||||
if cfg.Providers[1].GCP == nil || cfg.Providers[1].GCP.Project != "my-project" {
|
||||
t.Errorf("providers[1].gcp = %+v, want Project=my-project", cfg.Providers[1].GCP)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLoadConfig_invalid(t *testing.T) {
|
||||
t.Parallel()
|
||||
tests := []struct {
|
||||
name string
|
||||
yaml string
|
||||
wantErrSub string
|
||||
}{
|
||||
{
|
||||
name: "empty providers list",
|
||||
yaml: `providers: []`,
|
||||
wantErrSub: "at least one provider",
|
||||
},
|
||||
{
|
||||
name: "missing name",
|
||||
yaml: `
|
||||
providers:
|
||||
- type: kubernetes`,
|
||||
wantErrSub: "name is required",
|
||||
},
|
||||
{
|
||||
name: "duplicate name",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: kubernetes
|
||||
type: kubernetes
|
||||
- name: kubernetes
|
||||
type: kubernetes`,
|
||||
wantErrSub: "duplicate provider name",
|
||||
},
|
||||
{
|
||||
name: "unknown type",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1
|
||||
type: azure`,
|
||||
wantErrSub: `unknown provider type "azure"`,
|
||||
},
|
||||
{
|
||||
name: "missing type",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1`,
|
||||
wantErrSub: "type is required",
|
||||
},
|
||||
{
|
||||
name: "gcp missing project",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: gcp-eu
|
||||
type: gcp`,
|
||||
wantErrSub: "gcp.project is required",
|
||||
},
|
||||
{
|
||||
name: "gcp with empty project",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: gcp-eu
|
||||
type: gcp
|
||||
gcp:
|
||||
project: ""`,
|
||||
wantErrSub: "gcp.project is required",
|
||||
},
|
||||
{
|
||||
name: "kubernetes type with gcp block",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1
|
||||
type: kubernetes
|
||||
gcp:
|
||||
project: my-project`,
|
||||
wantErrSub: "type is kubernetes but a gcp block is set",
|
||||
},
|
||||
{
|
||||
name: "gcp type with kubernetes block",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1
|
||||
type: gcp
|
||||
gcp:
|
||||
project: my-project
|
||||
kubernetes:
|
||||
image: custom-image`,
|
||||
wantErrSub: "type is gcp but a kubernetes block is set",
|
||||
},
|
||||
{
|
||||
name: "strict mode rejects unknown top-level key",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1
|
||||
type: kubernetes
|
||||
extraneous: true`,
|
||||
wantErrSub: "parsing providers config",
|
||||
},
|
||||
{
|
||||
name: "strict mode rejects unknown provider key",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1
|
||||
type: kubernetes
|
||||
bogus: true`,
|
||||
wantErrSub: "parsing providers config",
|
||||
},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
_, err := LoadConfig([]byte(tc.yaml))
|
||||
if err == nil {
|
||||
t.Fatalf("LoadConfig() error = nil, want error containing %q", tc.wantErrSub)
|
||||
}
|
||||
if !strings.Contains(err.Error(), tc.wantErrSub) {
|
||||
t.Errorf("LoadConfig() error = %q, want substring %q", err.Error(), tc.wantErrSub)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestLoadConfigFile_missingFile(t *testing.T) {
|
||||
t.Parallel()
|
||||
_, err := LoadConfigFile("/nonexistent/providers.yaml")
|
||||
if err == nil {
|
||||
t.Fatal("LoadConfigFile() error = nil, want error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "reading providers config") {
|
||||
t.Errorf("LoadConfigFile() error = %q, want substring %q", err.Error(), "reading providers config")
|
||||
}
|
||||
}
|
||||
91
internal/provider/errors.go
Normal file
91
internal/provider/errors.go
Normal file
@@ -0,0 +1,91 @@
|
||||
package provider
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
)
|
||||
|
||||
// The provider error taxonomy. Every error a Provider method returns should
|
||||
// be classifiable as exactly one of these four, via Wrap or Class. The
|
||||
// reconciler branches on this classification to decide how to react —
|
||||
// never on a provider-specific error type.
|
||||
var (
|
||||
// ErrNotFound means the instance doesn't exist. Get returning this is
|
||||
// normal, not exceptional.
|
||||
ErrNotFound = errors.New("provider: instance not found")
|
||||
// ErrQuotaExceeded means the request failed because of a cloud quota or
|
||||
// rate limit. The reconciler backs off slowly (minutes, not seconds)
|
||||
// rather than hammering an exhausted quota.
|
||||
ErrQuotaExceeded = errors.New("provider: quota exceeded")
|
||||
// ErrTransient means the request failed for a reason likely to clear on
|
||||
// retry (network blip, 5xx, timeout). The reconciler retries with the
|
||||
// workqueue's normal exponential backoff.
|
||||
ErrTransient = errors.New("provider: transient error")
|
||||
// ErrPermanent means the request failed for a reason that will not
|
||||
// clear on retry (bad config, permission denied, invalid argument). The
|
||||
// reconciler stops retrying and surfaces the failure in status.
|
||||
ErrPermanent = errors.New("provider: permanent error")
|
||||
)
|
||||
|
||||
// Error wraps a provider-specific error with enough context for logs, while
|
||||
// remaining classifiable via errors.Is against one of the four sentinels
|
||||
// above and unwrappable via errors.As to the underlying SDK error.
|
||||
type Error struct {
|
||||
Class error // one of ErrNotFound/ErrQuotaExceeded/ErrTransient/ErrPermanent
|
||||
Op string // "create", "get", "delete", "list"
|
||||
Provider string // the configured provider name, e.g. "gcp-eu"
|
||||
ID string // providerID, if known
|
||||
Err error // underlying error; may be nil
|
||||
}
|
||||
|
||||
func (e *Error) Error() string {
|
||||
msg := fmt.Sprintf("provider %s: %s", e.Provider, e.Op)
|
||||
if e.ID != "" {
|
||||
msg += fmt.Sprintf(" %s", e.ID)
|
||||
}
|
||||
msg += ": " + e.Class.Error()
|
||||
if e.Err != nil {
|
||||
msg += fmt.Sprintf(" (%v)", e.Err)
|
||||
}
|
||||
return msg
|
||||
}
|
||||
|
||||
// Unwrap exposes both the taxonomy sentinel and the underlying error, via
|
||||
// Go's multi-error unwrap (errors.Unwrap() []error). This is what lets
|
||||
// errors.Is(err, ErrQuotaExceeded) and errors.As(err, &googleErr) both
|
||||
// succeed against the same *Error value: errors.Is/As walk every branch.
|
||||
func (e *Error) Unwrap() []error {
|
||||
if e.Err != nil {
|
||||
return []error{e.Class, e.Err}
|
||||
}
|
||||
return []error{e.Class}
|
||||
}
|
||||
|
||||
// Wrap builds an *Error classified under class. err is the underlying
|
||||
// SDK/network error and may be nil when there's nothing further to wrap
|
||||
// (e.g. classifying a bare HTTP status code).
|
||||
func Wrap(class error, op, providerName, id string, err error) error {
|
||||
return &Error{Class: class, Op: op, Provider: providerName, ID: id, Err: err}
|
||||
}
|
||||
|
||||
// Class returns the taxonomy sentinel matching err — ErrNotFound,
|
||||
// ErrQuotaExceeded, ErrPermanent, or ErrTransient, checked via errors.Is so
|
||||
// it works against any error built with Wrap regardless of how deeply it's
|
||||
// wrapped elsewhere. An unclassified error (nil, or not built with Wrap)
|
||||
// defaults to ErrTransient: retrying is always safer than latching Failed
|
||||
// on an error nobody has taught this package to recognize.
|
||||
func Class(err error) error {
|
||||
if err == nil {
|
||||
return nil
|
||||
}
|
||||
switch {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
return ErrNotFound
|
||||
case errors.Is(err, ErrQuotaExceeded):
|
||||
return ErrQuotaExceeded
|
||||
case errors.Is(err, ErrPermanent):
|
||||
return ErrPermanent
|
||||
default:
|
||||
return ErrTransient
|
||||
}
|
||||
}
|
||||
70
internal/provider/errors_test.go
Normal file
70
internal/provider/errors_test.go
Normal file
@@ -0,0 +1,70 @@
|
||||
package provider
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// sdkError simulates an underlying provider SDK error type, so tests can
|
||||
// assert errors.As reaches through the wrapper to it.
|
||||
type sdkError struct{ code int }
|
||||
|
||||
func (e *sdkError) Error() string { return fmt.Sprintf("sdk error %d", e.code) }
|
||||
|
||||
func TestError_IsAndAs(t *testing.T) {
|
||||
t.Parallel()
|
||||
underlying := &sdkError{code: 429}
|
||||
err := Wrap(ErrQuotaExceeded, "create", "gcp-eu", "zones/z/instances/x", underlying)
|
||||
|
||||
if !errors.Is(err, ErrQuotaExceeded) {
|
||||
t.Error("errors.Is(err, ErrQuotaExceeded) = false, want true")
|
||||
}
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
t.Error("errors.Is(err, ErrNotFound) = true, want false")
|
||||
}
|
||||
|
||||
var sdk *sdkError
|
||||
if !errors.As(err, &sdk) {
|
||||
t.Fatal("errors.As(err, &sdk) = false, want true")
|
||||
}
|
||||
if sdk.code != 429 {
|
||||
t.Errorf("recovered sdkError.code = %d, want 429", sdk.code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestError_WrapWithNilUnderlying(t *testing.T) {
|
||||
t.Parallel()
|
||||
err := Wrap(ErrPermanent, "get", "mock", "id", nil)
|
||||
if !errors.Is(err, ErrPermanent) {
|
||||
t.Error("errors.Is(err, ErrPermanent) = false, want true")
|
||||
}
|
||||
if err.Error() == "" {
|
||||
t.Error("Error() returned an empty string")
|
||||
}
|
||||
}
|
||||
|
||||
func TestClass(t *testing.T) {
|
||||
t.Parallel()
|
||||
tests := []struct {
|
||||
name string
|
||||
err error
|
||||
want error
|
||||
}{
|
||||
{"nil returns nil", nil, nil},
|
||||
{"wrapped not found", Wrap(ErrNotFound, "get", "mock", "id", nil), ErrNotFound},
|
||||
{"wrapped quota", Wrap(ErrQuotaExceeded, "create", "gcp", "id", nil), ErrQuotaExceeded},
|
||||
{"wrapped permanent", Wrap(ErrPermanent, "create", "gcp", "id", nil), ErrPermanent},
|
||||
{"wrapped transient", Wrap(ErrTransient, "create", "gcp", "id", nil), ErrTransient},
|
||||
{"unclassified defaults to transient", errors.New("boom"), ErrTransient},
|
||||
{"further-wrapped preserves classification", fmt.Errorf("outer: %w", Wrap(ErrQuotaExceeded, "create", "gcp", "id", nil)), ErrQuotaExceeded},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
if got := Class(tc.err); got != tc.want {
|
||||
t.Errorf("Class(%v) = %v, want %v", tc.err, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
81
internal/provider/gcp/errors.go
Normal file
81
internal/provider/gcp/errors.go
Normal file
@@ -0,0 +1,81 @@
|
||||
package gcp
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"net/http"
|
||||
"slices"
|
||||
|
||||
"github.com/go-logr/logr"
|
||||
"google.golang.org/api/googleapi"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// classify maps a GCP API error onto the provider taxonomy:
|
||||
// 404 → NotFound; 429 and quota-flavored 403s → QuotaExceeded;
|
||||
// 400/401/other 403s → Permanent; 408/5xx and anything unrecognized
|
||||
// (network errors, context cancellation) → Transient, because retrying is
|
||||
// always safer than latching Failed on an error nobody taught this
|
||||
// function to recognize.
|
||||
func classify(err error) error {
|
||||
var gerr *googleapi.Error
|
||||
if !errors.As(err, &gerr) {
|
||||
return provider.ErrTransient
|
||||
}
|
||||
switch {
|
||||
case gerr.Code == http.StatusNotFound:
|
||||
return provider.ErrNotFound
|
||||
case gerr.Code == http.StatusTooManyRequests:
|
||||
return provider.ErrQuotaExceeded
|
||||
case gerr.Code == http.StatusForbidden && hasReason(gerr, "quotaExceeded", "rateLimitExceeded"):
|
||||
return provider.ErrQuotaExceeded
|
||||
case gerr.Code == http.StatusBadRequest,
|
||||
gerr.Code == http.StatusUnauthorized,
|
||||
gerr.Code == http.StatusForbidden:
|
||||
return provider.ErrPermanent
|
||||
default:
|
||||
return provider.ErrTransient
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Provider) wrapErr(op, id string, err error) error {
|
||||
return provider.Wrap(classify(err), op, p.name, id, err)
|
||||
}
|
||||
|
||||
// logAPIError records the raw googleapi error shape (HTTP status, reasons)
|
||||
// at V(1) — classify collapses it onto the coarser provider taxonomy, so
|
||||
// this line is the only place the original status survives.
|
||||
func logAPIError(log logr.Logger, op string, err error) {
|
||||
if !log.V(1).Enabled() {
|
||||
return
|
||||
}
|
||||
kv := []any{"op", op, "error", err.Error()}
|
||||
var gerr *googleapi.Error
|
||||
if errors.As(err, &gerr) {
|
||||
reasons := make([]string, 0, len(gerr.Errors))
|
||||
for _, item := range gerr.Errors {
|
||||
reasons = append(reasons, item.Reason)
|
||||
}
|
||||
kv = append(kv, "httpStatus", gerr.Code, "reasons", reasons)
|
||||
}
|
||||
log.V(1).Info("GCP API call failed", kv...)
|
||||
}
|
||||
|
||||
func hasReason(gerr *googleapi.Error, reasons ...string) bool {
|
||||
for _, item := range gerr.Errors {
|
||||
if slices.Contains(reasons, item.Reason) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func isAlreadyExists(err error) bool {
|
||||
var gerr *googleapi.Error
|
||||
return errors.As(err, &gerr) && gerr.Code == http.StatusConflict
|
||||
}
|
||||
|
||||
func isNotFound(err error) bool {
|
||||
var gerr *googleapi.Error
|
||||
return errors.As(err, &gerr) && gerr.Code == http.StatusNotFound
|
||||
}
|
||||
71
internal/provider/gcp/errors_test.go
Normal file
71
internal/provider/gcp/errors_test.go
Normal file
@@ -0,0 +1,71 @@
|
||||
package gcp
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"testing"
|
||||
|
||||
"google.golang.org/api/googleapi"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
func gerr(code int, reasons ...string) error {
|
||||
e := &googleapi.Error{Code: code, Message: "boom"}
|
||||
for _, r := range reasons {
|
||||
e.Errors = append(e.Errors, googleapi.ErrorItem{Reason: r})
|
||||
}
|
||||
return e
|
||||
}
|
||||
|
||||
func TestClassify(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
err error
|
||||
want error
|
||||
}{
|
||||
{name: "404 is NotFound", err: gerr(404), want: provider.ErrNotFound},
|
||||
{name: "429 is Quota", err: gerr(429), want: provider.ErrQuotaExceeded},
|
||||
{name: "403 quotaExceeded is Quota", err: gerr(403, "quotaExceeded"), want: provider.ErrQuotaExceeded},
|
||||
{name: "403 rateLimitExceeded is Quota", err: gerr(403, "rateLimitExceeded"), want: provider.ErrQuotaExceeded},
|
||||
{name: "403 plain is Permanent", err: gerr(403, "forbidden"), want: provider.ErrPermanent},
|
||||
{name: "400 is Permanent", err: gerr(400), want: provider.ErrPermanent},
|
||||
{name: "401 is Permanent", err: gerr(401), want: provider.ErrPermanent},
|
||||
{name: "408 is Transient", err: gerr(408), want: provider.ErrTransient},
|
||||
{name: "500 is Transient", err: gerr(500), want: provider.ErrTransient},
|
||||
{name: "503 is Transient", err: gerr(503), want: provider.ErrTransient},
|
||||
{name: "409 is Transient (alreadyExists is handled before classify)", err: gerr(409), want: provider.ErrTransient},
|
||||
{name: "plain network error is Transient", err: errors.New("connection reset"), want: provider.ErrTransient},
|
||||
{name: "wrapped googleapi error still classifies", err: fmt.Errorf("calling api: %w", gerr(404)), want: provider.ErrNotFound},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
if got := classify(tc.err); got != tc.want {
|
||||
t.Errorf("classify() = %v, want %v", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The wrapped error must satisfy both halves of the taxonomy contract:
|
||||
// errors.Is against the sentinel AND errors.As back to the SDK error.
|
||||
func TestWrapErr_isAndAsBothWork(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := &Provider{name: "gcp-eu"}
|
||||
wrapped := p.wrapErr("get", "zones/z/instances/i", gerr(404))
|
||||
|
||||
if !errors.Is(wrapped, provider.ErrNotFound) {
|
||||
t.Error("errors.Is(wrapped, ErrNotFound) = false")
|
||||
}
|
||||
var ge *googleapi.Error
|
||||
if !errors.As(wrapped, &ge) || ge.Code != 404 {
|
||||
t.Error("errors.As back to *googleapi.Error failed")
|
||||
}
|
||||
if provider.Class(wrapped) != provider.ErrNotFound {
|
||||
t.Errorf("Class() = %v, want ErrNotFound", provider.Class(wrapped))
|
||||
}
|
||||
}
|
||||
307
internal/provider/gcp/gcp.go
Normal file
307
internal/provider/gcp/gcp.go
Normal file
@@ -0,0 +1,307 @@
|
||||
// Package gcp implements the provider contract on GCP Compute Engine via
|
||||
// the modern Cloud Client Library (cloud.google.com/go/compute/apiv1),
|
||||
// deliberately restricted to four calls: instances.Insert, Get, Delete,
|
||||
// AggregatedList. Operations are fire-and-forget — Operation.Wait is never
|
||||
// called; Create/Delete return as soon as the operation is submitted and
|
||||
// the reconciler discovers progress by polling Get.
|
||||
package gcp
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
compute "cloud.google.com/go/compute/apiv1"
|
||||
"cloud.google.com/go/compute/apiv1/computepb"
|
||||
"github.com/go-logr/logr"
|
||||
"google.golang.org/api/iterator"
|
||||
"google.golang.org/api/option"
|
||||
"google.golang.org/protobuf/proto"
|
||||
logf "sigs.k8s.io/controller-runtime/pkg/log"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// instancesAPI is the test seam. It deliberately does not mirror the SDK:
|
||||
// the SDK's InstancesScopedListPairIterator has an unexported nextFunc, so
|
||||
// a fake cannot construct one — the seam flattens AggregatedList to a
|
||||
// slice, and returns operations as just their name (the only thing this
|
||||
// provider ever uses, since it never waits on them).
|
||||
type instancesAPI interface {
|
||||
Insert(ctx context.Context, req *computepb.InsertInstanceRequest) (opName string, err error)
|
||||
Get(ctx context.Context, req *computepb.GetInstanceRequest) (*computepb.Instance, error)
|
||||
Delete(ctx context.Context, req *computepb.DeleteInstanceRequest) (opName string, err error)
|
||||
AggregatedList(ctx context.Context, req *computepb.AggregatedListInstancesRequest) ([]*computepb.Instance, error)
|
||||
}
|
||||
|
||||
// realInstances adapts *compute.InstancesClient to the seam.
|
||||
type realInstances struct {
|
||||
client *compute.InstancesClient
|
||||
}
|
||||
|
||||
func (r *realInstances) Insert(ctx context.Context, req *computepb.InsertInstanceRequest) (string, error) {
|
||||
op, err := r.client.Insert(ctx, req)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return op.Name(), nil
|
||||
}
|
||||
|
||||
func (r *realInstances) Get(ctx context.Context, req *computepb.GetInstanceRequest) (*computepb.Instance, error) {
|
||||
return r.client.Get(ctx, req)
|
||||
}
|
||||
|
||||
func (r *realInstances) Delete(ctx context.Context, req *computepb.DeleteInstanceRequest) (string, error) {
|
||||
op, err := r.client.Delete(ctx, req)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return op.Name(), nil
|
||||
}
|
||||
|
||||
func (r *realInstances) AggregatedList(ctx context.Context, req *computepb.AggregatedListInstancesRequest) ([]*computepb.Instance, error) {
|
||||
it := r.client.AggregatedList(ctx, req)
|
||||
var out []*computepb.Instance
|
||||
for {
|
||||
pair, err := it.Next()
|
||||
if err == iterator.Done {
|
||||
return out, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if pair.Value != nil {
|
||||
out = append(out, pair.Value.Instances...)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Provider implements provider.Provider on GCP Compute Engine.
|
||||
type Provider struct {
|
||||
name string
|
||||
cfg provider.GCPConfig
|
||||
api instancesAPI
|
||||
}
|
||||
|
||||
// New builds a Provider using Application Default Credentials (workload
|
||||
// identity in-cluster, gcloud ADC locally — no key-file plumbing).
|
||||
// Deliberately untested: it dials real Google endpoints; everything below
|
||||
// it is exercised through newWithAPI.
|
||||
func New(ctx context.Context, pc provider.ProviderConfig) (provider.Provider, error) {
|
||||
return NewWithWireOptions(ctx, pc, WireLogOptions{})
|
||||
}
|
||||
|
||||
// NewWithWireOptions is New with explicit control over the V(5) wire
|
||||
// logging; the injected wire logger surfaces the SDK's HTTP
|
||||
// request/response records at V(5). Note option.WithLogger overrides the
|
||||
// SDK's own GOOGLE_SDK_GO_LOGGING_LEVEL env var, so --zap-log-level is
|
||||
// the only knob.
|
||||
func NewWithWireOptions(ctx context.Context, pc provider.ProviderConfig, opts WireLogOptions) (provider.Provider, error) {
|
||||
base := logf.Log.WithName("gcp").WithName("http")
|
||||
if base.V(5).Enabled() {
|
||||
logf.Log.WithName("gcp").Info(
|
||||
"GCP HTTP wire logging active — request payloads include cloud-init user-data",
|
||||
"fullPayloads", opts.FullPayloads)
|
||||
}
|
||||
client, err := compute.NewInstancesRESTClient(ctx, option.WithLogger(wireLogger(base, opts)))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("creating GCP instances client: %w", err)
|
||||
}
|
||||
return newWithAPI(pc, &realInstances{client: client}), nil
|
||||
}
|
||||
|
||||
func newWithAPI(pc provider.ProviderConfig, api instancesAPI) *Provider {
|
||||
cfg := provider.GCPConfig{}
|
||||
if pc.GCP != nil {
|
||||
cfg = *pc.GCP
|
||||
}
|
||||
return &Provider{name: pc.Name, cfg: withDefaults(cfg), api: api}
|
||||
}
|
||||
|
||||
// logger derives the request-scoped logger from ctx, so provider lines
|
||||
// inherit the reconcile context (which Proxy triggered the call).
|
||||
func (p *Provider) logger(ctx context.Context) logr.Logger {
|
||||
return logf.FromContext(ctx).WithName("gcp").WithValues("provider", p.name)
|
||||
}
|
||||
|
||||
// Create submits the insert and returns immediately with the
|
||||
// zone-qualified providerID. A 409 alreadyExists is success — the
|
||||
// deterministic instance name means a repeat call after a crash found the
|
||||
// VM it already created, which is exactly the idempotency the contract
|
||||
// demands.
|
||||
func (p *Provider) Create(ctx context.Context, req provider.CreateRequest) (string, error) {
|
||||
pl := req.Placement
|
||||
if pl.Zone == "" || pl.MachineType == "" || pl.Image == "" {
|
||||
return "", provider.Wrap(provider.ErrPermanent, "create", p.name, "", fmt.Errorf(
|
||||
"gcp requires placement.zone, placement.machineType and placement.image (got zone=%q machineType=%q image=%q)",
|
||||
pl.Zone, pl.MachineType, pl.Image))
|
||||
}
|
||||
log := p.logger(ctx)
|
||||
id := formatProviderID(pl.Zone, req.Name)
|
||||
insertReq := buildInsertRequest(p.cfg, req)
|
||||
// Curated fields only: the request proto embeds the cloud-init
|
||||
// user-data, which may be Secret-sourced and must never reach logs.
|
||||
log.V(2).Info("GCP insert request built",
|
||||
"zone", pl.Zone, "name", req.Name,
|
||||
"network", p.cfg.Network, "networkTag", p.cfg.NetworkTag,
|
||||
"diskSizeGb", p.cfg.DiskSizeGB, "port", req.Port,
|
||||
"labels", insertReq.GetInstanceResource().GetLabels(),
|
||||
"cloudInitBytes", len(req.CloudInit))
|
||||
opName, err := p.api.Insert(ctx, insertReq)
|
||||
switch {
|
||||
case err == nil:
|
||||
log.V(1).Info("GCP instance insert submitted",
|
||||
"zone", pl.Zone, "name", req.Name,
|
||||
"machineType", pl.MachineType, "image", pl.Image, "opName", opName)
|
||||
case isAlreadyExists(err):
|
||||
log.V(1).Info("GCP instance already exists, insert treated as success",
|
||||
"zone", pl.Zone, "name", req.Name)
|
||||
default:
|
||||
logAPIError(log, "create", err)
|
||||
return "", p.wrapErr("create", id, err)
|
||||
}
|
||||
return id, nil
|
||||
}
|
||||
|
||||
// Get returns the instance state. The providerID carries its own zone, so
|
||||
// this stays correct even mid-replacement after a zone edit — re-reading
|
||||
// spec.placement.zone would look up the wrong zone exactly then.
|
||||
func (p *Provider) Get(ctx context.Context, providerID string) (*provider.Instance, error) {
|
||||
zone, name, err := parseProviderID(providerID)
|
||||
if err != nil {
|
||||
return nil, provider.Wrap(provider.ErrPermanent, "get", p.name, providerID, err)
|
||||
}
|
||||
log := p.logger(ctx)
|
||||
inst, err := p.api.Get(ctx, &computepb.GetInstanceRequest{
|
||||
Project: p.cfg.Project,
|
||||
Zone: zone,
|
||||
Instance: name,
|
||||
})
|
||||
if err != nil {
|
||||
logAPIError(log, "get", err)
|
||||
return nil, p.wrapErr("get", providerID, err)
|
||||
}
|
||||
out := toInstance(inst, zone)
|
||||
log.V(1).Info("GCP instance fetched",
|
||||
"zone", zone, "name", name,
|
||||
"status", inst.GetStatus(), "state", out.State, "ip", out.IP)
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Delete submits the delete and returns; deleting an instance that is
|
||||
// already gone is success.
|
||||
func (p *Provider) Delete(ctx context.Context, providerID string) error {
|
||||
zone, name, err := parseProviderID(providerID)
|
||||
if err != nil {
|
||||
return provider.Wrap(provider.ErrPermanent, "delete", p.name, providerID, err)
|
||||
}
|
||||
log := p.logger(ctx)
|
||||
opName, err := p.api.Delete(ctx, &computepb.DeleteInstanceRequest{
|
||||
Project: p.cfg.Project,
|
||||
Zone: zone,
|
||||
Instance: name,
|
||||
})
|
||||
switch {
|
||||
case err == nil:
|
||||
log.V(1).Info("GCP instance delete submitted",
|
||||
"zone", zone, "name", name, "opName", opName)
|
||||
case isNotFound(err):
|
||||
log.V(1).Info("GCP instance already gone, delete treated as success",
|
||||
"zone", zone, "name", name)
|
||||
default:
|
||||
logAPIError(log, "delete", err)
|
||||
return p.wrapErr("delete", providerID, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ListByTag sweeps every zone for instances carrying the GC labels.
|
||||
// ReturnPartialSuccess matters: without it one unreachable zone fails the
|
||||
// entire GC sweep.
|
||||
func (p *Provider) ListByTag(ctx context.Context) ([]provider.Instance, error) {
|
||||
log := p.logger(ctx)
|
||||
filter := fmt.Sprintf("labels.%s = %s", provider.LabelManaged, provider.LabelManagedYes)
|
||||
instances, err := p.api.AggregatedList(ctx, &computepb.AggregatedListInstancesRequest{
|
||||
Project: p.cfg.Project,
|
||||
Filter: proto.String(filter),
|
||||
ReturnPartialSuccess: proto.Bool(true),
|
||||
})
|
||||
if err != nil {
|
||||
logAPIError(log, "list", err)
|
||||
return nil, p.wrapErr("list", "", err)
|
||||
}
|
||||
log.V(1).Info("GCP instances listed", "filter", filter, "count", len(instances))
|
||||
out := make([]provider.Instance, 0, len(instances))
|
||||
for _, inst := range instances {
|
||||
conv := toInstance(inst, lastPathSegment(inst.GetZone()))
|
||||
out = append(out, *conv)
|
||||
log.V(2).Info("GCP listed instance",
|
||||
"id", conv.ID, "state", conv.State, "uid", conv.UID, "createdAt", conv.CreatedAt)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func toInstance(inst *computepb.Instance, zone string) *provider.Instance {
|
||||
var ip string
|
||||
if nics := inst.GetNetworkInterfaces(); len(nics) > 0 {
|
||||
if acs := nics[0].GetAccessConfigs(); len(acs) > 0 {
|
||||
ip = acs[0].GetNatIP()
|
||||
}
|
||||
}
|
||||
// CreationTimestamp is RFC3339; a parse failure leaves the zero time,
|
||||
// which orphan GC treats as "old" — safe, since a malformed timestamp
|
||||
// never protects a candidate from collection forever.
|
||||
created, _ := time.Parse(time.RFC3339, inst.GetCreationTimestamp())
|
||||
return &provider.Instance{
|
||||
ID: formatProviderID(zone, inst.GetName()),
|
||||
IP: ip,
|
||||
State: mapState(inst.GetStatus(), ip),
|
||||
UID: inst.GetLabels()[provider.LabelUID],
|
||||
CreatedAt: created,
|
||||
}
|
||||
}
|
||||
|
||||
// mapState collapses GCP instance statuses onto the provider states. A
|
||||
// RUNNING instance without a NatIP maps to Provisioning — an empty IP must
|
||||
// never be published as Running. Anything unrecognized maps to Stopped:
|
||||
// the reconciler's answer to Stopped is delete-and-recreate, which is
|
||||
// always safe for cattle.
|
||||
func mapState(status, ip string) provider.InstanceState {
|
||||
switch status {
|
||||
case "PROVISIONING", "STAGING", "REPAIRING":
|
||||
return provider.StateProvisioning
|
||||
case "RUNNING":
|
||||
if ip == "" {
|
||||
return provider.StateProvisioning
|
||||
}
|
||||
return provider.StateRunning
|
||||
case "STOPPING", "STOPPED", "SUSPENDING", "SUSPENDED":
|
||||
return provider.StateStopped
|
||||
case "TERMINATED":
|
||||
return provider.StateTerminated
|
||||
default:
|
||||
return provider.StateStopped
|
||||
}
|
||||
}
|
||||
|
||||
func formatProviderID(zone, name string) string {
|
||||
return fmt.Sprintf("zones/%s/instances/%s", zone, name)
|
||||
}
|
||||
|
||||
func parseProviderID(id string) (zone, name string, err error) {
|
||||
parts := strings.Split(id, "/")
|
||||
if len(parts) != 4 || parts[0] != "zones" || parts[2] != "instances" || parts[1] == "" || parts[3] == "" {
|
||||
return "", "", fmt.Errorf("malformed gcp providerID %q, want zones/<zone>/instances/<name>", id)
|
||||
}
|
||||
return parts[1], parts[3], nil
|
||||
}
|
||||
|
||||
// lastPathSegment extracts the zone name from the URL-style
|
||||
// ".../zones/europe-west1-b" the API returns on instances.
|
||||
func lastPathSegment(url string) string {
|
||||
if i := strings.LastIndexByte(url, '/'); i >= 0 {
|
||||
return url[i+1:]
|
||||
}
|
||||
return url
|
||||
}
|
||||
564
internal/provider/gcp/gcp_test.go
Normal file
564
internal/provider/gcp/gcp_test.go
Normal file
@@ -0,0 +1,564 @@
|
||||
package gcp
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"log/slog"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"cloud.google.com/go/compute/apiv1/computepb"
|
||||
"github.com/go-logr/logr"
|
||||
"github.com/go-logr/logr/funcr"
|
||||
"google.golang.org/protobuf/proto"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// fakeAPI implements the instancesAPI seam.
|
||||
type fakeAPI struct {
|
||||
insertReq *computepb.InsertInstanceRequest
|
||||
insertErr error
|
||||
|
||||
getReq *computepb.GetInstanceRequest
|
||||
getInst *computepb.Instance
|
||||
getErr error
|
||||
|
||||
deleteReq *computepb.DeleteInstanceRequest
|
||||
deleteErr error
|
||||
|
||||
listReq *computepb.AggregatedListInstancesRequest
|
||||
listInsts []*computepb.Instance
|
||||
listErr error
|
||||
}
|
||||
|
||||
func (f *fakeAPI) Insert(_ context.Context, req *computepb.InsertInstanceRequest) (string, error) {
|
||||
f.insertReq = req
|
||||
return "op-insert", f.insertErr
|
||||
}
|
||||
|
||||
func (f *fakeAPI) Get(_ context.Context, req *computepb.GetInstanceRequest) (*computepb.Instance, error) {
|
||||
f.getReq = req
|
||||
return f.getInst, f.getErr
|
||||
}
|
||||
|
||||
func (f *fakeAPI) Delete(_ context.Context, req *computepb.DeleteInstanceRequest) (string, error) {
|
||||
f.deleteReq = req
|
||||
return "op-delete", f.deleteErr
|
||||
}
|
||||
|
||||
func (f *fakeAPI) AggregatedList(_ context.Context, req *computepb.AggregatedListInstancesRequest) ([]*computepb.Instance, error) {
|
||||
f.listReq = req
|
||||
return f.listInsts, f.listErr
|
||||
}
|
||||
|
||||
func newTestProvider(api *fakeAPI) *Provider {
|
||||
return newWithAPI(provider.ProviderConfig{
|
||||
Name: "gcp-eu",
|
||||
Type: "gcp",
|
||||
GCP: &provider.GCPConfig{Project: "my-project"},
|
||||
}, api)
|
||||
}
|
||||
|
||||
func TestCreate_returnsZoneQualifiedID(t *testing.T) {
|
||||
t.Parallel()
|
||||
api := &fakeAPI{}
|
||||
p := newTestProvider(api)
|
||||
|
||||
id, err := p.Create(context.Background(), testCreateRequest())
|
||||
if err != nil {
|
||||
t.Fatalf("Create: %v", err)
|
||||
}
|
||||
if want := "zones/europe-west1-b/instances/proxy-abc123def456ghij"; id != want {
|
||||
t.Errorf("providerID = %s, want %s", id, want)
|
||||
}
|
||||
if api.insertReq.Project != "my-project" || api.insertReq.Zone != "europe-west1-b" {
|
||||
t.Errorf("insert sent to %s/%s, want my-project/europe-west1-b", api.insertReq.Project, api.insertReq.Zone)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCreate_alreadyExistsIsSuccess(t *testing.T) {
|
||||
t.Parallel()
|
||||
api := &fakeAPI{insertErr: gerr(409)}
|
||||
p := newTestProvider(api)
|
||||
|
||||
id, err := p.Create(context.Background(), testCreateRequest())
|
||||
if err != nil {
|
||||
t.Fatalf("Create after crash (409): %v — alreadyExists must be success", err)
|
||||
}
|
||||
if want := "zones/europe-west1-b/instances/proxy-abc123def456ghij"; id != want {
|
||||
t.Errorf("providerID = %s, want %s", id, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCreate_incompletePlacementIsPermanent(t *testing.T) {
|
||||
t.Parallel()
|
||||
api := &fakeAPI{}
|
||||
p := newTestProvider(api)
|
||||
req := testCreateRequest()
|
||||
req.Placement.MachineType = ""
|
||||
|
||||
_, err := p.Create(context.Background(), req)
|
||||
if provider.Class(err) != provider.ErrPermanent {
|
||||
t.Errorf("Class = %v, want ErrPermanent for missing placement", provider.Class(err))
|
||||
}
|
||||
if api.insertReq != nil {
|
||||
t.Error("Insert was called despite invalid placement")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCreate_quotaErrorClassified(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider(&fakeAPI{insertErr: gerr(403, "quotaExceeded")})
|
||||
_, err := p.Create(context.Background(), testCreateRequest())
|
||||
if provider.Class(err) != provider.ErrQuotaExceeded {
|
||||
t.Errorf("Class = %v, want ErrQuotaExceeded", provider.Class(err))
|
||||
}
|
||||
}
|
||||
|
||||
func TestGet_stateMapping(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
status string
|
||||
natIP string
|
||||
wantState provider.InstanceState
|
||||
wantIP string
|
||||
}{
|
||||
{name: "provisioning", status: "PROVISIONING", wantState: provider.StateProvisioning},
|
||||
{name: "staging", status: "STAGING", wantState: provider.StateProvisioning},
|
||||
{name: "repairing", status: "REPAIRING", wantState: provider.StateProvisioning},
|
||||
{name: "running without NatIP stays provisioning", status: "RUNNING", wantState: provider.StateProvisioning},
|
||||
{name: "running with NatIP", status: "RUNNING", natIP: "34.1.2.3", wantState: provider.StateRunning, wantIP: "34.1.2.3"},
|
||||
{name: "stopped", status: "STOPPED", wantState: provider.StateStopped},
|
||||
{name: "suspended", status: "SUSPENDED", wantState: provider.StateStopped},
|
||||
{name: "terminated", status: "TERMINATED", wantState: provider.StateTerminated},
|
||||
{name: "unknown status maps to stopped for recreation", status: "SOMETHING_NEW", wantState: provider.StateStopped},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
inst := &computepb.Instance{
|
||||
Name: proto.String("proxy-abc"),
|
||||
Status: proto.String(tc.status),
|
||||
CreationTimestamp: proto.String("2026-08-09T10:00:00+02:00"),
|
||||
Labels: map[string]string{
|
||||
provider.LabelUID: "uid-1",
|
||||
},
|
||||
}
|
||||
if tc.natIP != "" {
|
||||
inst.NetworkInterfaces = []*computepb.NetworkInterface{{
|
||||
AccessConfigs: []*computepb.AccessConfig{{NatIP: proto.String(tc.natIP)}},
|
||||
}}
|
||||
}
|
||||
p := newTestProvider(&fakeAPI{getInst: inst})
|
||||
|
||||
got, err := p.Get(context.Background(), "zones/europe-west1-b/instances/proxy-abc")
|
||||
if err != nil {
|
||||
t.Fatalf("Get: %v", err)
|
||||
}
|
||||
if got.State != tc.wantState || got.IP != tc.wantIP {
|
||||
t.Errorf("state/ip = %s/%q, want %s/%q", got.State, got.IP, tc.wantState, tc.wantIP)
|
||||
}
|
||||
if got.UID != "uid-1" {
|
||||
t.Errorf("UID = %q, want uid-1 (from the GC label)", got.UID)
|
||||
}
|
||||
if got.ID != "zones/europe-west1-b/instances/proxy-abc" {
|
||||
t.Errorf("ID = %s, want the zone-qualified providerID", got.ID)
|
||||
}
|
||||
if got.CreatedAt.IsZero() {
|
||||
t.Error("CreatedAt not parsed from creationTimestamp")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestGet_notFound(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider(&fakeAPI{getErr: gerr(404)})
|
||||
_, err := p.Get(context.Background(), "zones/z/instances/gone")
|
||||
if !errors.Is(err, provider.ErrNotFound) {
|
||||
t.Errorf("err = %v, want ErrNotFound", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestGet_malformedProviderID(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider(&fakeAPI{})
|
||||
for _, id := range []string{"", "proxy-abc", "zones//instances/x", "zones/z/instances/", "z/zone/i/name"} {
|
||||
if _, err := p.Get(context.Background(), id); provider.Class(err) != provider.ErrPermanent {
|
||||
t.Errorf("Get(%q): Class = %v, want ErrPermanent", id, provider.Class(err))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDelete_notFoundIsSuccess(t *testing.T) {
|
||||
t.Parallel()
|
||||
api := &fakeAPI{deleteErr: gerr(404)}
|
||||
p := newTestProvider(api)
|
||||
if err := p.Delete(context.Background(), "zones/z/instances/gone"); err != nil {
|
||||
t.Errorf("Delete of missing instance: %v, want nil", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDelete_sendsParsedZoneAndName(t *testing.T) {
|
||||
t.Parallel()
|
||||
api := &fakeAPI{}
|
||||
p := newTestProvider(api)
|
||||
if err := p.Delete(context.Background(), "zones/us-east1-c/instances/proxy-xyz"); err != nil {
|
||||
t.Fatalf("Delete: %v", err)
|
||||
}
|
||||
if api.deleteReq.Zone != "us-east1-c" || api.deleteReq.Instance != "proxy-xyz" {
|
||||
t.Errorf("delete sent %s/%s, want us-east1-c/proxy-xyz", api.deleteReq.Zone, api.deleteReq.Instance)
|
||||
}
|
||||
}
|
||||
|
||||
func TestListByTag(t *testing.T) {
|
||||
t.Parallel()
|
||||
api := &fakeAPI{listInsts: []*computepb.Instance{{
|
||||
Name: proto.String("proxy-old"),
|
||||
Status: proto.String("RUNNING"),
|
||||
Zone: proto.String("https://www.googleapis.com/compute/v1/projects/my-project/zones/europe-west1-b"),
|
||||
CreationTimestamp: proto.String(time.Now().Format(time.RFC3339)),
|
||||
Labels: map[string]string{
|
||||
provider.LabelManaged: provider.LabelManagedYes,
|
||||
provider.LabelUID: "uid-orphan",
|
||||
},
|
||||
NetworkInterfaces: []*computepb.NetworkInterface{{
|
||||
AccessConfigs: []*computepb.AccessConfig{{NatIP: proto.String("34.9.9.9")}},
|
||||
}},
|
||||
}}}
|
||||
p := newTestProvider(api)
|
||||
|
||||
got, err := p.ListByTag(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("ListByTag: %v", err)
|
||||
}
|
||||
|
||||
if want := "labels.proxy-operator-managed = true"; api.listReq.GetFilter() != want {
|
||||
t.Errorf("filter = %q, want %q", api.listReq.GetFilter(), want)
|
||||
}
|
||||
if !api.listReq.GetReturnPartialSuccess() {
|
||||
t.Error("ReturnPartialSuccess not set — one unreachable zone would fail the whole GC sweep")
|
||||
}
|
||||
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("instances = %d, want 1", len(got))
|
||||
}
|
||||
if got[0].ID != "zones/europe-west1-b/instances/proxy-old" {
|
||||
t.Errorf("ID = %s, want the zone parsed out of the URL-style zone field", got[0].ID)
|
||||
}
|
||||
if got[0].UID != "uid-orphan" || got[0].State != provider.StateRunning {
|
||||
t.Errorf("instance = %+v, want uid-orphan/Running", got[0])
|
||||
}
|
||||
}
|
||||
|
||||
func TestListByTag_errorPropagates(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider(&fakeAPI{listErr: gerr(500)})
|
||||
_, err := p.ListByTag(context.Background())
|
||||
if provider.Class(err) != provider.ErrTransient {
|
||||
t.Errorf("Class = %v, want ErrTransient", provider.Class(err))
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseProviderID_roundTrip(t *testing.T) {
|
||||
t.Parallel()
|
||||
id := formatProviderID("europe-west1-b", "proxy-abc")
|
||||
zone, name, err := parseProviderID(id)
|
||||
if err != nil || zone != "europe-west1-b" || name != "proxy-abc" {
|
||||
t.Errorf("round trip = %s/%s (%v), want europe-west1-b/proxy-abc", zone, name, err)
|
||||
}
|
||||
}
|
||||
|
||||
// captureContext returns a ctx carrying a funcr logger that records every
|
||||
// emitted line, capped at the given verbosity — the test stand-in for
|
||||
// --zap-log-level=<verbosity>.
|
||||
func captureContext(verbosity int) (context.Context, *[]string) {
|
||||
lines := &[]string{}
|
||||
log := funcr.New(func(prefix, args string) {
|
||||
*lines = append(*lines, prefix+" "+args)
|
||||
}, funcr.Options{Verbosity: verbosity})
|
||||
return logr.NewContext(context.Background(), log), lines
|
||||
}
|
||||
|
||||
func runningInstance() *computepb.Instance {
|
||||
return &computepb.Instance{
|
||||
Name: proto.String("proxy-abc123def456ghij"),
|
||||
Status: proto.String("RUNNING"),
|
||||
Zone: proto.String("https://www.googleapis.com/compute/v1/projects/my-project/zones/europe-west1-b"),
|
||||
CreationTimestamp: proto.String("2026-08-09T10:00:00+02:00"),
|
||||
Labels: map[string]string{
|
||||
provider.LabelManaged: provider.LabelManagedYes,
|
||||
provider.LabelUID: "uid-1",
|
||||
},
|
||||
NetworkInterfaces: []*computepb.NetworkInterface{{
|
||||
AccessConfigs: []*computepb.AccessConfig{{NatIP: proto.String("34.1.2.3")}},
|
||||
}},
|
||||
}
|
||||
}
|
||||
|
||||
func runAllOps(t *testing.T, ctx context.Context, req provider.CreateRequest) {
|
||||
t.Helper()
|
||||
inst := runningInstance()
|
||||
p := newTestProvider(&fakeAPI{getInst: inst, listInsts: []*computepb.Instance{inst}})
|
||||
if _, err := p.Create(ctx, req); err != nil {
|
||||
t.Fatalf("Create: %v", err)
|
||||
}
|
||||
if _, err := p.Get(ctx, "zones/europe-west1-b/instances/proxy-abc123def456ghij"); err != nil {
|
||||
t.Fatalf("Get: %v", err)
|
||||
}
|
||||
if err := p.Delete(ctx, "zones/europe-west1-b/instances/proxy-abc123def456ghij"); err != nil {
|
||||
t.Fatalf("Delete: %v", err)
|
||||
}
|
||||
if _, err := p.ListByTag(ctx); err != nil {
|
||||
t.Fatalf("ListByTag: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLogging_verbosityTiers(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
verbosity int
|
||||
wantLines []string
|
||||
absentLines []string
|
||||
}{
|
||||
{
|
||||
name: "v0 stays silent",
|
||||
verbosity: 0,
|
||||
absentLines: []string{
|
||||
"GCP instance insert submitted",
|
||||
"GCP instance fetched",
|
||||
"GCP instance delete submitted",
|
||||
"GCP instances listed",
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "v1 logs one line per API call",
|
||||
verbosity: 1,
|
||||
wantLines: []string{
|
||||
`"msg"="GCP instance insert submitted"`,
|
||||
`"opName"="op-insert"`,
|
||||
`"msg"="GCP instance fetched"`,
|
||||
`"status"="RUNNING"`,
|
||||
`"msg"="GCP instance delete submitted"`,
|
||||
`"opName"="op-delete"`,
|
||||
`"msg"="GCP instances listed"`,
|
||||
`"provider"="gcp-eu"`,
|
||||
},
|
||||
absentLines: []string{
|
||||
"GCP insert request built",
|
||||
"GCP listed instance",
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "v2 adds request and per-instance detail",
|
||||
verbosity: 2,
|
||||
wantLines: []string{
|
||||
`"msg"="GCP insert request built"`,
|
||||
`"cloudInitBytes"=`,
|
||||
`"msg"="GCP listed instance"`,
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
ctx, lines := captureContext(tc.verbosity)
|
||||
req := testCreateRequest()
|
||||
const sentinel = "SENTINEL-cloud-init-must-never-be-logged"
|
||||
req.CloudInit = sentinel
|
||||
|
||||
runAllOps(t, ctx, req)
|
||||
|
||||
joined := strings.Join(*lines, "\n")
|
||||
if tc.verbosity == 0 && len(*lines) != 0 {
|
||||
t.Errorf("verbosity 0 logged %d lines:\n%s", len(*lines), joined)
|
||||
}
|
||||
for _, want := range tc.wantLines {
|
||||
if !strings.Contains(joined, want) {
|
||||
t.Errorf("output missing %q:\n%s", want, joined)
|
||||
}
|
||||
}
|
||||
for _, absent := range tc.absentLines {
|
||||
if strings.Contains(joined, absent) {
|
||||
t.Errorf("output unexpectedly contains %q:\n%s", absent, joined)
|
||||
}
|
||||
}
|
||||
if strings.Contains(joined, sentinel) {
|
||||
t.Errorf("cloud-init content leaked into logs:\n%s", joined)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestWireLogger_gatesAtV5(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
verbosity int
|
||||
wantDebug bool
|
||||
}{
|
||||
{name: "v5 shows wire records", verbosity: 5, wantDebug: true},
|
||||
{name: "v4 hides wire records", verbosity: 4, wantDebug: false},
|
||||
{name: "v2 hides wire records", verbosity: 2, wantDebug: false},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
lines := &[]string{}
|
||||
base := funcr.New(func(prefix, args string) {
|
||||
*lines = append(*lines, prefix+" "+args)
|
||||
}, funcr.Options{Verbosity: tc.verbosity})
|
||||
|
||||
slogger := wireLogger(base, WireLogOptions{})
|
||||
slogger.Debug("api request", "rpcName", "Insert")
|
||||
|
||||
joined := strings.Join(*lines, "\n")
|
||||
if got := strings.Contains(joined, "api request"); got != tc.wantDebug {
|
||||
t.Errorf("Debug record visible = %v, want %v; output:\n%s", got, tc.wantDebug, joined)
|
||||
}
|
||||
if tc.wantDebug && !strings.Contains(joined, "rpcName") {
|
||||
t.Errorf("wire record lost its attrs:\n%s", joined)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestWireLogger_infoLandsAtV1(t *testing.T) {
|
||||
t.Parallel()
|
||||
lines := &[]string{}
|
||||
base := funcr.New(func(prefix, args string) {
|
||||
*lines = append(*lines, prefix+" "+args)
|
||||
}, funcr.Options{Verbosity: 1})
|
||||
|
||||
wireLogger(base, WireLogOptions{}).Info("hello")
|
||||
|
||||
if joined := strings.Join(*lines, "\n"); !strings.Contains(joined, "hello") {
|
||||
t.Errorf("slog Info should land at V(1) and be visible at verbosity 1; output:\n%s", joined)
|
||||
}
|
||||
}
|
||||
|
||||
func captureWireLogger(verbosity int, opts WireLogOptions) (*slog.Logger, *[]string) {
|
||||
lines := &[]string{}
|
||||
base := funcr.New(func(prefix, args string) {
|
||||
*lines = append(*lines, prefix+" "+args)
|
||||
}, funcr.Options{Verbosity: verbosity})
|
||||
return wireLogger(base, opts), lines
|
||||
}
|
||||
|
||||
func TestWireLogger_dropsNonAPIDebugRecords(t *testing.T) {
|
||||
t.Parallel()
|
||||
slogger, lines := captureWireLogger(9, WireLogOptions{})
|
||||
|
||||
const secret = "assertion=eyJhbGciOiJSUzI1NiJ9.SECRET"
|
||||
slogger.Debug("2LO token request", "request", map[string]any{"payload": secret})
|
||||
slogger.Debug("2LO token response", "response", map[string]any{"payload": "ya29.SECRET-TOKEN"})
|
||||
|
||||
if len(*lines) != 0 {
|
||||
t.Errorf("auth token-exchange records must be dropped; got:\n%s", strings.Join(*lines, "\n"))
|
||||
}
|
||||
|
||||
slogger.Warn("credential refresh failed")
|
||||
if joined := strings.Join(*lines, "\n"); !strings.Contains(joined, "credential refresh failed") {
|
||||
t.Errorf("non-debug SDK records should pass through; output:\n%s", joined)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWireLogger_elidesLargeFields(t *testing.T) {
|
||||
t.Parallel()
|
||||
slogger, lines := captureWireLogger(9, WireLogOptions{})
|
||||
|
||||
huge := strings.Repeat("x", 4096)
|
||||
slogger.Debug("api response", "response", map[string]any{
|
||||
"status": "200",
|
||||
"payload": map[string]any{
|
||||
"name": "proxy-abc",
|
||||
"disks": []any{map[string]any{"content": huge}},
|
||||
},
|
||||
})
|
||||
|
||||
joined := strings.Join(*lines, "\n")
|
||||
if strings.Contains(joined, huge[:64]) {
|
||||
t.Errorf("large field not elided:\n%.500s", joined)
|
||||
}
|
||||
if !strings.Contains(joined, "[elided 4096 bytes]") {
|
||||
t.Errorf("elision marker missing:\n%s", joined)
|
||||
}
|
||||
for _, keep := range []string{"proxy-abc", "200", "api response"} {
|
||||
if !strings.Contains(joined, keep) {
|
||||
t.Errorf("small field %q lost during elision:\n%s", keep, joined)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestWireLogger_fullPayloadsDisablesElision(t *testing.T) {
|
||||
t.Parallel()
|
||||
slogger, lines := captureWireLogger(9, WireLogOptions{FullPayloads: true})
|
||||
|
||||
huge := strings.Repeat("y", 4096)
|
||||
slogger.Debug("api response", "response", map[string]any{"payload": huge})
|
||||
|
||||
joined := strings.Join(*lines, "\n")
|
||||
if !strings.Contains(joined, huge) {
|
||||
t.Errorf("FullPayloads should keep fields verbatim:\n%.200s", joined)
|
||||
}
|
||||
|
||||
slogger.Debug("2LO token response", "response", "ya29.SECRET")
|
||||
if joined := strings.Join(*lines, "\n"); strings.Contains(joined, "ya29.SECRET") {
|
||||
t.Error("auth records must be dropped even with FullPayloads")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLogging_apiErrorKeepsHTTPDetail(t *testing.T) {
|
||||
t.Parallel()
|
||||
ctx, lines := captureContext(1)
|
||||
p := newTestProvider(&fakeAPI{insertErr: gerr(403, "quotaExceeded")})
|
||||
|
||||
if _, err := p.Create(ctx, testCreateRequest()); err == nil {
|
||||
t.Fatal("Create: want error")
|
||||
}
|
||||
|
||||
joined := strings.Join(*lines, "\n")
|
||||
for _, want := range []string{
|
||||
`"msg"="GCP API call failed"`,
|
||||
`"httpStatus"=403`,
|
||||
`"quotaExceeded"`,
|
||||
`"op"="create"`,
|
||||
} {
|
||||
if !strings.Contains(joined, want) {
|
||||
t.Errorf("output missing %q:\n%s", want, joined)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestLogging_treatedAsSuccessPathsAreExplicit(t *testing.T) {
|
||||
t.Parallel()
|
||||
ctx, lines := captureContext(1)
|
||||
p := newTestProvider(&fakeAPI{insertErr: gerr(409), deleteErr: gerr(404)})
|
||||
|
||||
if _, err := p.Create(ctx, testCreateRequest()); err != nil {
|
||||
t.Fatalf("Create with 409: %v", err)
|
||||
}
|
||||
if err := p.Delete(ctx, "zones/z/instances/gone"); err != nil {
|
||||
t.Fatalf("Delete with 404: %v", err)
|
||||
}
|
||||
|
||||
joined := strings.Join(*lines, "\n")
|
||||
for _, want := range []string{
|
||||
`"msg"="GCP instance already exists, insert treated as success"`,
|
||||
`"msg"="GCP instance already gone, delete treated as success"`,
|
||||
} {
|
||||
if !strings.Contains(joined, want) {
|
||||
t.Errorf("output missing %q:\n%s", want, joined)
|
||||
}
|
||||
}
|
||||
}
|
||||
74
internal/provider/gcp/insert.go
Normal file
74
internal/provider/gcp/insert.go
Normal file
@@ -0,0 +1,74 @@
|
||||
package gcp
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"cloud.google.com/go/compute/apiv1/computepb"
|
||||
"google.golang.org/protobuf/proto"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
const (
|
||||
defaultNetwork = "default"
|
||||
defaultNetworkTag = "proxy-operator"
|
||||
defaultDiskSizeGB = 10
|
||||
userDataKey = "user-data"
|
||||
)
|
||||
|
||||
func withDefaults(cfg provider.GCPConfig) provider.GCPConfig {
|
||||
if cfg.Network == "" {
|
||||
cfg.Network = defaultNetwork
|
||||
}
|
||||
if cfg.NetworkTag == "" {
|
||||
cfg.NetworkTag = defaultNetworkTag
|
||||
}
|
||||
if cfg.DiskSizeGB == 0 {
|
||||
cfg.DiskSizeGB = defaultDiskSizeGB
|
||||
}
|
||||
return cfg
|
||||
}
|
||||
|
||||
// buildInsertRequest is pure so the field-by-field unit test needs no fake
|
||||
// at all — the plan's primary test for this provider.
|
||||
func buildInsertRequest(cfg provider.GCPConfig, req provider.CreateRequest) *computepb.InsertInstanceRequest {
|
||||
inst := &computepb.Instance{
|
||||
Name: proto.String(req.Name),
|
||||
MachineType: proto.String(fmt.Sprintf("zones/%s/machineTypes/%s", req.Placement.Zone, req.Placement.MachineType)),
|
||||
Disks: []*computepb.AttachedDisk{{
|
||||
Boot: proto.Bool(true),
|
||||
AutoDelete: proto.Bool(true),
|
||||
InitializeParams: &computepb.AttachedDiskInitializeParams{
|
||||
SourceImage: proto.String(req.Placement.Image),
|
||||
DiskSizeGb: proto.Int64(cfg.DiskSizeGB),
|
||||
},
|
||||
}},
|
||||
NetworkInterfaces: []*computepb.NetworkInterface{{
|
||||
Network: proto.String("global/networks/" + cfg.Network),
|
||||
// An ephemeral external IP: exactly this pair, per the API's
|
||||
// contract for one-to-one NAT.
|
||||
AccessConfigs: []*computepb.AccessConfig{{
|
||||
Name: proto.String("External NAT"),
|
||||
Type: proto.String("ONE_TO_ONE_NAT"),
|
||||
}},
|
||||
}},
|
||||
// The GC contract: every resource this operator creates carries
|
||||
// these two labels, and orphan GC relies on both.
|
||||
Labels: map[string]string{
|
||||
provider.LabelManaged: provider.LabelManagedYes,
|
||||
provider.LabelUID: req.UID,
|
||||
},
|
||||
Tags: &computepb.Tags{Items: []string{cfg.NetworkTag}},
|
||||
}
|
||||
if req.CloudInit != "" {
|
||||
inst.Metadata = &computepb.Metadata{Items: []*computepb.Items{{
|
||||
Key: proto.String(userDataKey),
|
||||
Value: proto.String(req.CloudInit),
|
||||
}}}
|
||||
}
|
||||
return &computepb.InsertInstanceRequest{
|
||||
Project: cfg.Project,
|
||||
Zone: req.Placement.Zone,
|
||||
InstanceResource: inst,
|
||||
}
|
||||
}
|
||||
127
internal/provider/gcp/insert_test.go
Normal file
127
internal/provider/gcp/insert_test.go
Normal file
@@ -0,0 +1,127 @@
|
||||
package gcp
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
func testCreateRequest() provider.CreateRequest {
|
||||
return provider.CreateRequest{
|
||||
Name: "proxy-abc123def456ghij",
|
||||
UID: "11111111-2222-3333-4444-555555555555",
|
||||
Namespace: "default",
|
||||
ProxyName: "eu-proxy-1",
|
||||
Placement: provider.Placement{
|
||||
Zone: "europe-west1-b",
|
||||
MachineType: "e2-micro",
|
||||
Image: "projects/debian-cloud/global/images/family/debian-12",
|
||||
},
|
||||
CloudInit: "#cloud-config\npackages: [squid]",
|
||||
Port: 3128,
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildInsertRequest_fieldByField(t *testing.T) {
|
||||
t.Parallel()
|
||||
cfg := withDefaults(provider.GCPConfig{Project: "my-project"})
|
||||
req := buildInsertRequest(cfg, testCreateRequest())
|
||||
|
||||
if req.Project != "my-project" || req.Zone != "europe-west1-b" {
|
||||
t.Errorf("project/zone = %s/%s, want my-project/europe-west1-b", req.Project, req.Zone)
|
||||
}
|
||||
inst := req.InstanceResource
|
||||
if inst.GetName() != "proxy-abc123def456ghij" {
|
||||
t.Errorf("name = %s", inst.GetName())
|
||||
}
|
||||
if got, want := inst.GetMachineType(), "zones/europe-west1-b/machineTypes/e2-micro"; got != want {
|
||||
t.Errorf("machineType = %s, want %s", got, want)
|
||||
}
|
||||
|
||||
if len(inst.GetDisks()) != 1 {
|
||||
t.Fatalf("disks = %d, want 1", len(inst.GetDisks()))
|
||||
}
|
||||
disk := inst.GetDisks()[0]
|
||||
if !disk.GetBoot() || !disk.GetAutoDelete() {
|
||||
t.Errorf("boot/autoDelete = %v/%v, want true/true", disk.GetBoot(), disk.GetAutoDelete())
|
||||
}
|
||||
if got, want := disk.GetInitializeParams().GetSourceImage(), "projects/debian-cloud/global/images/family/debian-12"; got != want {
|
||||
t.Errorf("sourceImage = %s, want %s", got, want)
|
||||
}
|
||||
if disk.GetInitializeParams().GetDiskSizeGb() != 10 {
|
||||
t.Errorf("diskSizeGb = %d, want the 10 default", disk.GetInitializeParams().GetDiskSizeGb())
|
||||
}
|
||||
|
||||
if len(inst.GetNetworkInterfaces()) != 1 {
|
||||
t.Fatalf("networkInterfaces = %d, want 1", len(inst.GetNetworkInterfaces()))
|
||||
}
|
||||
nic := inst.GetNetworkInterfaces()[0]
|
||||
if got, want := nic.GetNetwork(), "global/networks/default"; got != want {
|
||||
t.Errorf("network = %s, want %s", got, want)
|
||||
}
|
||||
if len(nic.GetAccessConfigs()) != 1 {
|
||||
t.Fatalf("accessConfigs = %d, want 1", len(nic.GetAccessConfigs()))
|
||||
}
|
||||
ac := nic.GetAccessConfigs()[0]
|
||||
if ac.GetName() != "External NAT" || ac.GetType() != "ONE_TO_ONE_NAT" {
|
||||
t.Errorf("accessConfig = %s/%s, want External NAT/ONE_TO_ONE_NAT", ac.GetName(), ac.GetType())
|
||||
}
|
||||
|
||||
wantLabels := map[string]string{
|
||||
provider.LabelManaged: provider.LabelManagedYes,
|
||||
provider.LabelUID: "11111111-2222-3333-4444-555555555555",
|
||||
}
|
||||
labels := inst.GetLabels()
|
||||
if len(labels) != len(wantLabels) {
|
||||
t.Errorf("labels = %v, want %v", labels, wantLabels)
|
||||
}
|
||||
for k, v := range wantLabels {
|
||||
if labels[k] != v {
|
||||
t.Errorf("label %s = %q, want %q", k, labels[k], v)
|
||||
}
|
||||
}
|
||||
|
||||
if tags := inst.GetTags().GetItems(); len(tags) != 1 || tags[0] != "proxy-operator" {
|
||||
t.Errorf("tags = %v, want [proxy-operator]", tags)
|
||||
}
|
||||
|
||||
items := inst.GetMetadata().GetItems()
|
||||
if len(items) != 1 || items[0].GetKey() != "user-data" {
|
||||
t.Fatalf("metadata items = %v, want one user-data entry", items)
|
||||
}
|
||||
if items[0].GetValue() != "#cloud-config\npackages: [squid]" {
|
||||
t.Errorf("user-data = %q, want the resolved cloud-init", items[0].GetValue())
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildInsertRequest_configOverrides(t *testing.T) {
|
||||
t.Parallel()
|
||||
cfg := withDefaults(provider.GCPConfig{
|
||||
Project: "my-project",
|
||||
Network: "crawl-vpc",
|
||||
NetworkTag: "crawl-egress",
|
||||
DiskSizeGB: 42,
|
||||
})
|
||||
req := buildInsertRequest(cfg, testCreateRequest())
|
||||
inst := req.InstanceResource
|
||||
|
||||
if got, want := inst.GetNetworkInterfaces()[0].GetNetwork(), "global/networks/crawl-vpc"; got != want {
|
||||
t.Errorf("network = %s, want %s", got, want)
|
||||
}
|
||||
if tags := inst.GetTags().GetItems(); len(tags) != 1 || tags[0] != "crawl-egress" {
|
||||
t.Errorf("tags = %v, want [crawl-egress]", tags)
|
||||
}
|
||||
if got := inst.GetDisks()[0].GetInitializeParams().GetDiskSizeGb(); got != 42 {
|
||||
t.Errorf("diskSizeGb = %d, want 42", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildInsertRequest_noCloudInitMeansNoMetadata(t *testing.T) {
|
||||
t.Parallel()
|
||||
req := testCreateRequest()
|
||||
req.CloudInit = ""
|
||||
built := buildInsertRequest(withDefaults(provider.GCPConfig{Project: "p"}), req)
|
||||
if built.InstanceResource.GetMetadata() != nil {
|
||||
t.Errorf("metadata = %v, want none without cloud-init", built.InstanceResource.GetMetadata())
|
||||
}
|
||||
}
|
||||
117
internal/provider/gcp/wirelog.go
Normal file
117
internal/provider/gcp/wirelog.go
Normal file
@@ -0,0 +1,117 @@
|
||||
package gcp
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
|
||||
"github.com/go-logr/logr"
|
||||
)
|
||||
|
||||
// wireLogMaxFieldBytes is the elision threshold for string fields in wire
|
||||
// payloads: GCP responses embed multi-KB blobs (Shielded-VM UEFI dbx
|
||||
// databases, licenses) that swamp the log line without diagnostic value.
|
||||
const wireLogMaxFieldBytes = 1024
|
||||
|
||||
// WireLogOptions controls the V(5) HTTP wire logging of the GCP SDK.
|
||||
type WireLogOptions struct {
|
||||
// FullPayloads disables field elision and logs payloads verbatim.
|
||||
FullPayloads bool
|
||||
}
|
||||
|
||||
// wireLogger returns the slog logger handed to the SDK: its Debug-level
|
||||
// "api request"/"api response" records (slog Debug = +4 on the logr
|
||||
// scale) land at V(5) on top of the base's V(1) shift.
|
||||
//
|
||||
// Debug records other than the compute client's api request/response are
|
||||
// dropped entirely: the same logger propagates into the auth library,
|
||||
// whose token-exchange records contain the signed JWT assertion and the
|
||||
// bearer access token. Warnings and errors pass through.
|
||||
func wireLogger(base logr.Logger, opts WireLogOptions) *slog.Logger {
|
||||
return slog.New(&wireFilterHandler{
|
||||
inner: logr.ToSlogHandler(base.V(1)),
|
||||
fullPayloads: opts.FullPayloads,
|
||||
})
|
||||
}
|
||||
|
||||
type wireFilterHandler struct {
|
||||
inner slog.Handler
|
||||
fullPayloads bool
|
||||
}
|
||||
|
||||
func (h *wireFilterHandler) Enabled(ctx context.Context, level slog.Level) bool {
|
||||
return h.inner.Enabled(ctx, level)
|
||||
}
|
||||
|
||||
func (h *wireFilterHandler) Handle(ctx context.Context, rec slog.Record) error {
|
||||
if rec.Level <= slog.LevelDebug && rec.Message != "api request" && rec.Message != "api response" {
|
||||
return nil
|
||||
}
|
||||
if h.fullPayloads {
|
||||
return h.inner.Handle(ctx, rec)
|
||||
}
|
||||
elided := slog.NewRecord(rec.Time, rec.Level, rec.Message, rec.PC)
|
||||
rec.Attrs(func(a slog.Attr) bool {
|
||||
elided.AddAttrs(slog.Attr{Key: a.Key, Value: elideValue(a.Value)})
|
||||
return true
|
||||
})
|
||||
return h.inner.Handle(ctx, elided)
|
||||
}
|
||||
|
||||
func (h *wireFilterHandler) WithAttrs(attrs []slog.Attr) slog.Handler {
|
||||
return &wireFilterHandler{inner: h.inner.WithAttrs(attrs), fullPayloads: h.fullPayloads}
|
||||
}
|
||||
|
||||
func (h *wireFilterHandler) WithGroup(name string) slog.Handler {
|
||||
return &wireFilterHandler{inner: h.inner.WithGroup(name), fullPayloads: h.fullPayloads}
|
||||
}
|
||||
|
||||
func elideValue(v slog.Value) slog.Value {
|
||||
v = v.Resolve()
|
||||
switch v.Kind() {
|
||||
case slog.KindString:
|
||||
if s := v.String(); len(s) > wireLogMaxFieldBytes {
|
||||
return slog.StringValue(elisionMarker(len(s)))
|
||||
}
|
||||
return v
|
||||
case slog.KindGroup:
|
||||
attrs := v.Group()
|
||||
out := make([]slog.Attr, 0, len(attrs))
|
||||
for _, a := range attrs {
|
||||
out = append(out, slog.Attr{Key: a.Key, Value: elideValue(a.Value)})
|
||||
}
|
||||
return slog.GroupValue(out...)
|
||||
case slog.KindAny:
|
||||
return slog.AnyValue(elideAny(v.Any()))
|
||||
default:
|
||||
return v
|
||||
}
|
||||
}
|
||||
|
||||
func elideAny(v any) any {
|
||||
switch t := v.(type) {
|
||||
case string:
|
||||
if len(t) > wireLogMaxFieldBytes {
|
||||
return elisionMarker(len(t))
|
||||
}
|
||||
return t
|
||||
case map[string]any:
|
||||
out := make(map[string]any, len(t))
|
||||
for k, val := range t {
|
||||
out[k] = elideAny(val)
|
||||
}
|
||||
return out
|
||||
case []any:
|
||||
out := make([]any, len(t))
|
||||
for i, val := range t {
|
||||
out[i] = elideAny(val)
|
||||
}
|
||||
return out
|
||||
default:
|
||||
return v
|
||||
}
|
||||
}
|
||||
|
||||
func elisionMarker(size int) string {
|
||||
return fmt.Sprintf("[elided %d bytes]", size)
|
||||
}
|
||||
202
internal/provider/kubernetes/kubernetes.go
Normal file
202
internal/provider/kubernetes/kubernetes.go
Normal file
@@ -0,0 +1,202 @@
|
||||
// Package kubernetes is a provider.Provider that creates real Pods running
|
||||
// a Squid container in the same cluster the operator itself runs in —
|
||||
// unlike a cloud provider, "creating compute" here means talking back to
|
||||
// the very Kubernetes API the operator is already watching.
|
||||
package kubernetes
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"k8s.io/client-go/tools/cache"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// defaultImage is Canonical's actively maintained Squid image for Ubuntu
|
||||
// 24.04 LTS, verified before picking it: a public image (no build/load
|
||||
// step needed for a kind demo), 50M+ pulls, updated the same day this
|
||||
// decision was made.
|
||||
const defaultImage = "ubuntu/squid:6.6-24.04_edge"
|
||||
|
||||
// Provider creates proxy Pods. It builds its own client rather than
|
||||
// depending on the manager's, via ctrl.GetConfig() — which auto-detects
|
||||
// in-cluster config when running as a Pod and falls back to the local
|
||||
// kubeconfig otherwise. That's what makes `make run` against a local kind
|
||||
// cluster and running in-cluster use the exact same code path with no
|
||||
// provider-specific wiring in cmd/main.go.
|
||||
type Provider struct {
|
||||
client client.Client
|
||||
image string
|
||||
}
|
||||
|
||||
// New builds a kubernetes Provider from its config block. Satisfies
|
||||
// registry.Constructor.
|
||||
func New(_ context.Context, cfg provider.ProviderConfig) (provider.Provider, error) {
|
||||
restCfg, err := ctrl.GetConfig()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("kubernetes provider %q: loading kubeconfig: %w", cfg.Name, err)
|
||||
}
|
||||
scheme := runtime.NewScheme()
|
||||
if err := corev1.AddToScheme(scheme); err != nil {
|
||||
return nil, fmt.Errorf("kubernetes provider %q: %w", cfg.Name, err)
|
||||
}
|
||||
c, err := client.New(restCfg, client.Options{Scheme: scheme})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("kubernetes provider %q: building client: %w", cfg.Name, err)
|
||||
}
|
||||
return newWithClient(c, cfg), nil
|
||||
}
|
||||
|
||||
// newWithClient builds a Provider around an already-constructed client,
|
||||
// bypassing ctrl.GetConfig(). Tests use this exclusively — New() must
|
||||
// never run under `go test`, since ctrl.GetConfig() would happily connect
|
||||
// to whatever real cluster the developer's kubeconfig points at.
|
||||
func newWithClient(c client.Client, cfg provider.ProviderConfig) *Provider {
|
||||
image := defaultImage
|
||||
if cfg.Kubernetes != nil && cfg.Kubernetes.Image != "" {
|
||||
image = cfg.Kubernetes.Image
|
||||
}
|
||||
return &Provider{client: c, image: image}
|
||||
}
|
||||
|
||||
// Create creates a Pod running the proxy container. Idempotent by
|
||||
// req.Name: if a Pod with that name already exists in req.Namespace, its
|
||||
// providerID is returned rather than erroring, so a repeat call after a
|
||||
// crash finds the existing Pod instead of creating a duplicate.
|
||||
func (p *Provider) Create(ctx context.Context, req provider.CreateRequest) (string, error) {
|
||||
pod := buildPod(p.image, req)
|
||||
if err := p.client.Create(ctx, pod); err != nil {
|
||||
if apierrors.IsAlreadyExists(err) {
|
||||
return providerID(req.Namespace, req.Name), nil
|
||||
}
|
||||
return "", classify("create", req.Name, err)
|
||||
}
|
||||
return providerID(req.Namespace, req.Name), nil
|
||||
}
|
||||
|
||||
// Get returns the current state of a previously created Pod.
|
||||
func (p *Provider) Get(ctx context.Context, id string) (*provider.Instance, error) {
|
||||
ns, name, err := parseProviderID(id)
|
||||
if err != nil {
|
||||
return nil, provider.Wrap(provider.ErrPermanent, "get", "kubernetes", id, err)
|
||||
}
|
||||
var pod corev1.Pod
|
||||
if err := p.client.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &pod); err != nil {
|
||||
if apierrors.IsNotFound(err) {
|
||||
return nil, provider.Wrap(provider.ErrNotFound, "get", "kubernetes", id, nil)
|
||||
}
|
||||
return nil, classify("get", id, err)
|
||||
}
|
||||
return instanceFromPod(&pod), nil
|
||||
}
|
||||
|
||||
// Delete is idempotent: deleting an unknown Pod is not an error.
|
||||
func (p *Provider) Delete(ctx context.Context, id string) error {
|
||||
ns, name, err := parseProviderID(id)
|
||||
if err != nil {
|
||||
return provider.Wrap(provider.ErrPermanent, "delete", "kubernetes", id, err)
|
||||
}
|
||||
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name}}
|
||||
if err := p.client.Delete(ctx, pod); err != nil && !apierrors.IsNotFound(err) {
|
||||
return classify("delete", id, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ListByTag returns every Pod carrying LabelManaged, across every
|
||||
// namespace — orphan GC needs to find proxy Pods regardless of which
|
||||
// namespace Proxy CRs happen to live in, which is why this provider's RBAC
|
||||
// is cluster-scoped rather than namespaced.
|
||||
func (p *Provider) ListByTag(ctx context.Context) ([]provider.Instance, error) {
|
||||
var pods corev1.PodList
|
||||
if err := p.client.List(ctx, &pods, client.MatchingLabels{
|
||||
provider.LabelManaged: provider.LabelManagedYes,
|
||||
}); err != nil {
|
||||
return nil, classify("list", "", err)
|
||||
}
|
||||
out := make([]provider.Instance, 0, len(pods.Items))
|
||||
for i := range pods.Items {
|
||||
out = append(out, *instanceFromPod(&pods.Items[i]))
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// providerID encodes namespace and name into the opaque providerID string
|
||||
// the Provider interface exposes, so Get/Delete are self-contained without
|
||||
// needing to separately track or re-derive which namespace a Pod lives in
|
||||
// — the same reasoning as the GCP provider's zone-qualified providerID.
|
||||
func providerID(namespace, name string) string {
|
||||
return namespace + "/" + name
|
||||
}
|
||||
|
||||
func parseProviderID(id string) (namespace, name string, err error) {
|
||||
ns, n, err := cache.SplitMetaNamespaceKey(id)
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
if ns == "" {
|
||||
return "", "", fmt.Errorf("providerID %q missing a namespace", id)
|
||||
}
|
||||
return ns, n, nil
|
||||
}
|
||||
|
||||
func instanceFromPod(pod *corev1.Pod) *provider.Instance {
|
||||
inst := &provider.Instance{
|
||||
ID: providerID(pod.Namespace, pod.Name),
|
||||
UID: pod.Labels[provider.LabelUID],
|
||||
CreatedAt: pod.CreationTimestamp.Time,
|
||||
State: stateFromPod(pod),
|
||||
}
|
||||
if inst.State == provider.StateRunning {
|
||||
inst.IP = pod.Status.PodIP
|
||||
}
|
||||
return inst
|
||||
}
|
||||
|
||||
// stateFromPod maps a Pod's phase to InstanceState. Succeeded/Failed/
|
||||
// Unknown all collapse to Terminated: the reconciler treats Stopped and
|
||||
// Terminated identically (delete and recreate — cattle, not pets), so a
|
||||
// finer-grained distinction between "the container exited" and "the node
|
||||
// went unreachable" wouldn't change any behavior.
|
||||
func stateFromPod(pod *corev1.Pod) provider.InstanceState {
|
||||
switch pod.Status.Phase {
|
||||
case corev1.PodPending:
|
||||
return provider.StateProvisioning
|
||||
case corev1.PodRunning:
|
||||
if pod.Status.PodIP == "" {
|
||||
// Never publish an empty IP while the kubelet is still
|
||||
// finishing setup.
|
||||
return provider.StateProvisioning
|
||||
}
|
||||
return provider.StateRunning
|
||||
default: // Succeeded, Failed, Unknown
|
||||
return provider.StateTerminated
|
||||
}
|
||||
}
|
||||
|
||||
// classify maps a Kubernetes API error to the provider error taxonomy.
|
||||
// Quota-exceeded and RBAC-denied both surface as 403 Forbidden from the
|
||||
// API server with nothing in apierrors to tell them apart programmatically
|
||||
// — a known simplification for this prototype; both classify as
|
||||
// ErrPermanent, which is the safer default of the two (stop retrying
|
||||
// rather than hammering an API server that will never allow the request).
|
||||
func classify(op, id string, err error) error {
|
||||
switch {
|
||||
case apierrors.IsNotFound(err):
|
||||
return provider.Wrap(provider.ErrNotFound, op, "kubernetes", id, err)
|
||||
case apierrors.IsTooManyRequests(err), apierrors.IsServerTimeout(err), apierrors.IsTimeout(err):
|
||||
return provider.Wrap(provider.ErrTransient, op, "kubernetes", id, err)
|
||||
case apierrors.IsForbidden(err), apierrors.IsInvalid(err), apierrors.IsBadRequest(err), apierrors.IsUnauthorized(err):
|
||||
return provider.Wrap(provider.ErrPermanent, op, "kubernetes", id, err)
|
||||
default:
|
||||
return provider.Wrap(provider.ErrTransient, op, "kubernetes", id, err)
|
||||
}
|
||||
}
|
||||
245
internal/provider/kubernetes/kubernetes_test.go
Normal file
245
internal/provider/kubernetes/kubernetes_test.go
Normal file
@@ -0,0 +1,245 @@
|
||||
package kubernetes
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/runtime/schema"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
func newTestProvider(objs ...runtime.Object) *Provider {
|
||||
scheme := runtime.NewScheme()
|
||||
if err := corev1.AddToScheme(scheme); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
c := fake.NewClientBuilder().WithScheme(scheme).WithRuntimeObjects(objs...).Build()
|
||||
return newWithClient(c, provider.ProviderConfig{Name: "kubernetes"})
|
||||
}
|
||||
|
||||
func testPod(namespace, name, uid string, phase corev1.PodPhase, ip string) *corev1.Pod {
|
||||
return &corev1.Pod{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: name,
|
||||
Namespace: namespace,
|
||||
Labels: map[string]string{
|
||||
provider.LabelManaged: provider.LabelManagedYes,
|
||||
provider.LabelUID: uid,
|
||||
},
|
||||
},
|
||||
Status: corev1.PodStatus{Phase: phase, PodIP: ip},
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Create_buildsPodAndReturnsProviderID(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider()
|
||||
id, err := p.Create(context.Background(), provider.CreateRequest{
|
||||
Name: "proxy-abc", UID: "uid-1", Namespace: "crawl", Port: 3128,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Create() error = %v", err)
|
||||
}
|
||||
if id != "crawl/proxy-abc" {
|
||||
t.Errorf("Create() id = %q, want %q", id, "crawl/proxy-abc")
|
||||
}
|
||||
|
||||
var pod corev1.Pod
|
||||
key := types.NamespacedName{Namespace: "crawl", Name: "proxy-abc"}
|
||||
if err := p.client.Get(context.Background(), key, &pod); err != nil {
|
||||
t.Fatalf("expected the Pod to exist: %v", err)
|
||||
}
|
||||
if pod.Labels[provider.LabelUID] != "uid-1" {
|
||||
t.Errorf("pod UID label = %q, want %q", pod.Labels[provider.LabelUID], "uid-1")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Create_idempotent(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider()
|
||||
ctx := context.Background()
|
||||
req := provider.CreateRequest{Name: "proxy-dup", UID: "uid-2", Namespace: "crawl", Port: 3128}
|
||||
|
||||
id1, err := p.Create(ctx, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Create() #1 error = %v", err)
|
||||
}
|
||||
id2, err := p.Create(ctx, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Create() #2 error = %v, want nil (AlreadyExists must be absorbed)", err)
|
||||
}
|
||||
if id1 != id2 {
|
||||
t.Errorf("Create() not idempotent: %q != %q", id1, id2)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Get_stateMapping(t *testing.T) {
|
||||
t.Parallel()
|
||||
tests := []struct {
|
||||
name string
|
||||
phase corev1.PodPhase
|
||||
ip string
|
||||
wantState provider.InstanceState
|
||||
wantIP string
|
||||
}{
|
||||
{"pending", corev1.PodPending, "", provider.StateProvisioning, ""},
|
||||
{"running with IP", corev1.PodRunning, "10.0.0.5", provider.StateRunning, "10.0.0.5"},
|
||||
{"running without IP yet", corev1.PodRunning, "", provider.StateProvisioning, ""},
|
||||
{"succeeded", corev1.PodSucceeded, "10.0.0.5", provider.StateTerminated, ""},
|
||||
{"failed", corev1.PodFailed, "10.0.0.5", provider.StateTerminated, ""},
|
||||
{"unknown", corev1.PodUnknown, "10.0.0.5", provider.StateTerminated, ""},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
pod := testPod("crawl", "proxy-"+tc.name, "uid", tc.phase, tc.ip)
|
||||
p := newTestProvider(pod)
|
||||
inst, err := p.Get(context.Background(), "crawl/"+pod.Name)
|
||||
if err != nil {
|
||||
t.Fatalf("Get() error = %v", err)
|
||||
}
|
||||
if inst.State != tc.wantState {
|
||||
t.Errorf("State = %v, want %v", inst.State, tc.wantState)
|
||||
}
|
||||
if inst.IP != tc.wantIP {
|
||||
t.Errorf("IP = %q, want %q", inst.IP, tc.wantIP)
|
||||
}
|
||||
if inst.UID != "uid" {
|
||||
t.Errorf("UID = %q, want %q", inst.UID, "uid")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Get_notFound(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider()
|
||||
_, err := p.Get(context.Background(), "crawl/does-not-exist")
|
||||
if !errors.Is(err, provider.ErrNotFound) {
|
||||
t.Errorf("Get() error = %v, want ErrNotFound", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Get_malformedProviderID(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider()
|
||||
_, err := p.Get(context.Background(), "no-namespace-here")
|
||||
if err == nil {
|
||||
t.Fatal("Get() error = nil, want error for a providerID with no namespace")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Delete_idempotent(t *testing.T) {
|
||||
t.Parallel()
|
||||
pod := testPod("crawl", "proxy-del", "uid", corev1.PodRunning, "10.0.0.5")
|
||||
p := newTestProvider(pod)
|
||||
ctx := context.Background()
|
||||
|
||||
if err := p.Delete(ctx, "crawl/proxy-del"); err != nil {
|
||||
t.Fatalf("Delete() error = %v", err)
|
||||
}
|
||||
if err := p.Delete(ctx, "crawl/proxy-del"); err != nil {
|
||||
t.Errorf("Delete() (repeat) error = %v, want nil", err)
|
||||
}
|
||||
if err := p.Delete(ctx, "crawl/never-existed"); err != nil {
|
||||
t.Errorf("Delete() on unknown ID error = %v, want nil", err)
|
||||
}
|
||||
|
||||
if _, err := p.Get(ctx, "crawl/proxy-del"); !errors.Is(err, provider.ErrNotFound) {
|
||||
t.Errorf("Get() after Delete() error = %v, want ErrNotFound", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Delete_malformedProviderID(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider()
|
||||
if err := p.Delete(context.Background(), "no-namespace-here"); err == nil {
|
||||
t.Fatal("Delete() error = nil, want error for a providerID with no namespace")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_ListByTag_filtersByLabelAcrossNamespaces(t *testing.T) {
|
||||
t.Parallel()
|
||||
managed1 := testPod("crawl", "proxy-a", "uid-a", corev1.PodRunning, "10.0.0.1")
|
||||
managed2 := testPod("other-ns", "proxy-b", "uid-b", corev1.PodPending, "")
|
||||
unmanaged := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "not-ours", Namespace: "crawl"}}
|
||||
|
||||
p := newTestProvider(managed1, managed2, unmanaged)
|
||||
instances, err := p.ListByTag(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("ListByTag() error = %v", err)
|
||||
}
|
||||
if len(instances) != 2 {
|
||||
t.Fatalf("len(instances) = %d, want 2 (unmanaged Pod must be excluded)", len(instances))
|
||||
}
|
||||
|
||||
byID := make(map[string]provider.Instance, len(instances))
|
||||
for _, inst := range instances {
|
||||
byID[inst.ID] = inst
|
||||
}
|
||||
a, ok := byID["crawl/proxy-a"]
|
||||
if !ok {
|
||||
t.Fatal(`instances missing "crawl/proxy-a"`)
|
||||
}
|
||||
if a.State != provider.StateRunning || a.IP != "10.0.0.1" {
|
||||
t.Errorf("crawl/proxy-a = %+v, want Running/10.0.0.1", a)
|
||||
}
|
||||
if _, ok := byID["other-ns/proxy-b"]; !ok {
|
||||
t.Fatal(`instances missing "other-ns/proxy-b" (ListByTag must not be namespace-scoped)`)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewWithClient_image(t *testing.T) {
|
||||
t.Parallel()
|
||||
scheme := runtime.NewScheme()
|
||||
_ = corev1.AddToScheme(scheme)
|
||||
c := fake.NewClientBuilder().WithScheme(scheme).Build()
|
||||
|
||||
def := newWithClient(c, provider.ProviderConfig{Name: "k8s"})
|
||||
if def.image != defaultImage {
|
||||
t.Errorf("default image = %q, want %q", def.image, defaultImage)
|
||||
}
|
||||
|
||||
custom := newWithClient(c, provider.ProviderConfig{
|
||||
Name: "k8s",
|
||||
Kubernetes: &provider.KubernetesConfig{Image: "myregistry/squid:custom"},
|
||||
})
|
||||
if custom.image != "myregistry/squid:custom" {
|
||||
t.Errorf("custom image = %q, want %q", custom.image, "myregistry/squid:custom")
|
||||
}
|
||||
}
|
||||
|
||||
func TestClassify(t *testing.T) {
|
||||
t.Parallel()
|
||||
gr := schema.GroupResource{Group: "", Resource: "pods"}
|
||||
tests := []struct {
|
||||
name string
|
||||
err error
|
||||
want error
|
||||
}{
|
||||
{"not found", apierrors.NewNotFound(gr, "x"), provider.ErrNotFound},
|
||||
{"too many requests", apierrors.NewTooManyRequests("slow down", 5), provider.ErrTransient},
|
||||
{"server timeout", apierrors.NewServerTimeout(gr, "create", 5), provider.ErrTransient},
|
||||
{"forbidden", apierrors.NewForbidden(gr, "x", errors.New("denied")), provider.ErrPermanent},
|
||||
{"bad request", apierrors.NewBadRequest("bad"), provider.ErrPermanent},
|
||||
{"unauthorized", apierrors.NewUnauthorized("no creds"), provider.ErrPermanent},
|
||||
{"unclassified", errors.New("boom"), provider.ErrTransient},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
got := classify("op", "id", tc.err)
|
||||
if !errors.Is(got, tc.want) {
|
||||
t.Errorf("classify(%v) = %v, want class %v", tc.err, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
75
internal/provider/kubernetes/pod.go
Normal file
75
internal/provider/kubernetes/pod.go
Normal file
@@ -0,0 +1,75 @@
|
||||
package kubernetes
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
const proxyContainerName = "squid"
|
||||
|
||||
// writeConfigAndExec is the container's entrypoint: write the config the
|
||||
// SQUID_CONF env var carries to disk, then exec squid against it. Avoids a
|
||||
// separate ConfigMap object per proxy instance — there's still only one
|
||||
// Kubernetes object (the Pod) to create, track, and clean up per proxy.
|
||||
const writeConfigAndExec = `printf '%s' "$SQUID_CONF" > /etc/squid/squid.conf && exec squid -N -f /etc/squid/squid.conf`
|
||||
|
||||
// buildPod constructs the Pod for a proxy instance. Pure and side-effect
|
||||
// free, so it's unit-tested directly without a cluster — the same pattern
|
||||
// the GCP provider's buildInsertRequest uses.
|
||||
func buildPod(image string, req provider.CreateRequest) *corev1.Pod {
|
||||
return &corev1.Pod{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: req.Name,
|
||||
Namespace: req.Namespace,
|
||||
Labels: map[string]string{
|
||||
provider.LabelManaged: provider.LabelManagedYes,
|
||||
provider.LabelUID: req.UID,
|
||||
},
|
||||
},
|
||||
Spec: corev1.PodSpec{
|
||||
RestartPolicy: corev1.RestartPolicyAlways,
|
||||
Containers: []corev1.Container{{
|
||||
Name: proxyContainerName,
|
||||
Image: image,
|
||||
Command: []string{"/bin/sh", "-c"},
|
||||
Args: []string{writeConfigAndExec},
|
||||
Env: []corev1.EnvVar{{
|
||||
Name: "SQUID_CONF",
|
||||
Value: squidConf(req.Port),
|
||||
}},
|
||||
Ports: []corev1.ContainerPort{{
|
||||
ContainerPort: req.Port,
|
||||
Protocol: corev1.ProtocolTCP,
|
||||
}},
|
||||
}},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// squidConf generates a minimal Squid config listening on port, permissive
|
||||
// enough to forward CONNECT and plain HTTP to any destination. Open by
|
||||
// design: this is a proxy for a private cluster, not internet-facing, and
|
||||
// installing/configuring proxy software is explicitly out of scope for
|
||||
// this operator's real (GCP) provider too — cloud-init is passed through
|
||||
// there, never interpreted. via/forwarded_for are turned off so the proxy
|
||||
// doesn't leak the Pod's identity to the origin.
|
||||
func squidConf(port int32) string {
|
||||
// max_filedescriptors is load-bearing in containers: squid sizes its FD
|
||||
// tables from RLIMIT_NOFILE at startup, and containerd commonly sets
|
||||
// that to effectively unlimited (kind: ~10^9) — squid then allocates
|
||||
// gigabytes and is OOM-killed before it ever listens. cache_mem is
|
||||
// trimmed because a forwarding proxy for crawling gains nothing from
|
||||
// squid's 256 MB default cache.
|
||||
return fmt.Sprintf(`http_port %d
|
||||
acl all src 0.0.0.0/0
|
||||
http_access allow all
|
||||
via off
|
||||
forwarded_for off
|
||||
max_filedescriptors 1024
|
||||
cache_mem 16 MB
|
||||
`, port)
|
||||
}
|
||||
94
internal/provider/kubernetes/pod_test.go
Normal file
94
internal/provider/kubernetes/pod_test.go
Normal file
@@ -0,0 +1,94 @@
|
||||
package kubernetes
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
func TestBuildPod(t *testing.T) {
|
||||
t.Parallel()
|
||||
req := provider.CreateRequest{
|
||||
Name: "proxy-abc123",
|
||||
UID: "uid-1",
|
||||
Namespace: "crawl",
|
||||
ProxyName: "proxy-eu-1",
|
||||
Port: 3128,
|
||||
}
|
||||
pod := buildPod("ubuntu/squid:6.6-24.04_edge", req)
|
||||
|
||||
if pod.Name != req.Name {
|
||||
t.Errorf("pod.Name = %q, want %q", pod.Name, req.Name)
|
||||
}
|
||||
if pod.Namespace != req.Namespace {
|
||||
t.Errorf("pod.Namespace = %q, want %q", pod.Namespace, req.Namespace)
|
||||
}
|
||||
if pod.Labels[provider.LabelManaged] != provider.LabelManagedYes {
|
||||
t.Errorf("labels[%s] = %q, want %q", provider.LabelManaged, pod.Labels[provider.LabelManaged], provider.LabelManagedYes)
|
||||
}
|
||||
if pod.Labels[provider.LabelUID] != req.UID {
|
||||
t.Errorf("labels[%s] = %q, want %q", provider.LabelUID, pod.Labels[provider.LabelUID], req.UID)
|
||||
}
|
||||
if pod.Spec.RestartPolicy != corev1.RestartPolicyAlways {
|
||||
t.Errorf("RestartPolicy = %v, want Always", pod.Spec.RestartPolicy)
|
||||
}
|
||||
if len(pod.Spec.Containers) != 1 {
|
||||
t.Fatalf("len(Containers) = %d, want 1", len(pod.Spec.Containers))
|
||||
}
|
||||
c := pod.Spec.Containers[0]
|
||||
if c.Image != "ubuntu/squid:6.6-24.04_edge" {
|
||||
t.Errorf("Image = %q, want ubuntu/squid:6.6-24.04_edge", c.Image)
|
||||
}
|
||||
if len(c.Ports) != 1 || c.Ports[0].ContainerPort != req.Port {
|
||||
t.Errorf("Ports = %+v, want a single entry on port %d", c.Ports, req.Port)
|
||||
}
|
||||
var confEnv string
|
||||
for _, e := range c.Env {
|
||||
if e.Name == "SQUID_CONF" {
|
||||
confEnv = e.Value
|
||||
}
|
||||
}
|
||||
if !strings.Contains(confEnv, "http_port 3128") {
|
||||
t.Errorf("SQUID_CONF env = %q, want it to contain %q", confEnv, "http_port 3128")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildPod_usesRequestPort(t *testing.T) {
|
||||
t.Parallel()
|
||||
req := provider.CreateRequest{Name: "proxy-x", UID: "uid-2", Namespace: "ns", Port: 8080}
|
||||
pod := buildPod("img", req)
|
||||
conf := envValue(t, pod, "SQUID_CONF")
|
||||
if !strings.Contains(conf, "http_port 8080") {
|
||||
t.Errorf("SQUID_CONF = %q, want it to contain %q", conf, "http_port 8080")
|
||||
}
|
||||
if pod.Spec.Containers[0].Ports[0].ContainerPort != 8080 {
|
||||
t.Errorf("ContainerPort = %d, want 8080", pod.Spec.Containers[0].Ports[0].ContainerPort)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSquidConf_permissive(t *testing.T) {
|
||||
t.Parallel()
|
||||
conf := squidConf(3128)
|
||||
// max_filedescriptors guards against squid sizing its FD tables from a
|
||||
// container's effectively-unlimited RLIMIT_NOFILE and getting OOM-killed
|
||||
// at startup — found by the kind verification run, must not regress.
|
||||
for _, want := range []string{"http_port 3128", "http_access allow all", "via off", "forwarded_for off", "max_filedescriptors 1024"} {
|
||||
if !strings.Contains(conf, want) {
|
||||
t.Errorf("squidConf() = %q, want it to contain %q", conf, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func envValue(t *testing.T, pod *corev1.Pod, name string) string {
|
||||
t.Helper()
|
||||
for _, e := range pod.Spec.Containers[0].Env {
|
||||
if e.Name == name {
|
||||
return e.Value
|
||||
}
|
||||
}
|
||||
t.Fatalf("env var %q not found on container", name)
|
||||
return ""
|
||||
}
|
||||
66
internal/provider/metrics.go
Normal file
66
internal/provider/metrics.go
Normal file
@@ -0,0 +1,66 @@
|
||||
package provider
|
||||
|
||||
import "context"
|
||||
|
||||
// RequestRecorder receives one record per provider API call. Implemented
|
||||
// by internal/metrics; defined here so this package needs no metrics
|
||||
// dependency.
|
||||
type RequestRecorder interface {
|
||||
ProviderRequest(provider, op, result string)
|
||||
}
|
||||
|
||||
// WithMetrics wraps a Provider so every call is recorded with its
|
||||
// classified result — zero-cost instrumentation for the next five
|
||||
// providers, and the one place Class is called purely for observability.
|
||||
func WithMetrics(name string, p Provider, rec RequestRecorder) Provider {
|
||||
return &instrumented{name: name, inner: p, rec: rec}
|
||||
}
|
||||
|
||||
type instrumented struct {
|
||||
name string
|
||||
inner Provider
|
||||
rec RequestRecorder
|
||||
}
|
||||
|
||||
func (i *instrumented) Create(ctx context.Context, req CreateRequest) (string, error) {
|
||||
id, err := i.inner.Create(ctx, req)
|
||||
i.record("create", err)
|
||||
return id, err
|
||||
}
|
||||
|
||||
func (i *instrumented) Get(ctx context.Context, providerID string) (*Instance, error) {
|
||||
inst, err := i.inner.Get(ctx, providerID)
|
||||
i.record("get", err)
|
||||
return inst, err
|
||||
}
|
||||
|
||||
func (i *instrumented) Delete(ctx context.Context, providerID string) error {
|
||||
err := i.inner.Delete(ctx, providerID)
|
||||
i.record("delete", err)
|
||||
return err
|
||||
}
|
||||
|
||||
func (i *instrumented) ListByTag(ctx context.Context) ([]Instance, error) {
|
||||
instances, err := i.inner.ListByTag(ctx)
|
||||
i.record("list", err)
|
||||
return instances, err
|
||||
}
|
||||
|
||||
func (i *instrumented) record(op string, err error) {
|
||||
i.rec.ProviderRequest(i.name, op, resultLabel(err))
|
||||
}
|
||||
|
||||
func resultLabel(err error) string {
|
||||
switch Class(err) {
|
||||
case nil:
|
||||
return "ok"
|
||||
case ErrNotFound:
|
||||
return "not_found"
|
||||
case ErrQuotaExceeded:
|
||||
return "quota_exceeded"
|
||||
case ErrPermanent:
|
||||
return "permanent"
|
||||
default:
|
||||
return "transient"
|
||||
}
|
||||
}
|
||||
100
internal/provider/metrics_test.go
Normal file
100
internal/provider/metrics_test.go
Normal file
@@ -0,0 +1,100 @@
|
||||
package provider
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
)
|
||||
|
||||
type recordedCall struct{ provider, op, result string }
|
||||
|
||||
type fakeRecorder struct{ calls []recordedCall }
|
||||
|
||||
func (f *fakeRecorder) ProviderRequest(provider, op, result string) {
|
||||
f.calls = append(f.calls, recordedCall{provider, op, result})
|
||||
}
|
||||
|
||||
// staticProvider returns canned values; only the classification of its
|
||||
// errors matters here.
|
||||
type staticProvider struct {
|
||||
createErr, deleteErr, getErr, listErr error
|
||||
}
|
||||
|
||||
func (s *staticProvider) Create(context.Context, CreateRequest) (string, error) {
|
||||
return "id-1", s.createErr
|
||||
}
|
||||
func (s *staticProvider) Get(context.Context, string) (*Instance, error) {
|
||||
return &Instance{ID: "id-1"}, s.getErr
|
||||
}
|
||||
func (s *staticProvider) Delete(context.Context, string) error { return s.deleteErr }
|
||||
func (s *staticProvider) ListByTag(context.Context) ([]Instance, error) { return nil, s.listErr }
|
||||
|
||||
func TestWithMetrics_recordsClassifiedResults(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
inner *staticProvider
|
||||
call func(p Provider) error
|
||||
wantOp string
|
||||
wantResult string
|
||||
}{
|
||||
{
|
||||
name: "successful create is ok",
|
||||
inner: &staticProvider{},
|
||||
call: func(p Provider) error { _, err := p.Create(context.Background(), CreateRequest{}); return err },
|
||||
wantOp: "create",
|
||||
wantResult: "ok",
|
||||
},
|
||||
{
|
||||
name: "get NotFound",
|
||||
inner: &staticProvider{getErr: Wrap(ErrNotFound, "get", "x", "id-1", nil)},
|
||||
call: func(p Provider) error { _, err := p.Get(context.Background(), "id-1"); return err },
|
||||
wantOp: "get",
|
||||
wantResult: "not_found",
|
||||
},
|
||||
{
|
||||
name: "create quota",
|
||||
inner: &staticProvider{createErr: Wrap(ErrQuotaExceeded, "create", "x", "", nil)},
|
||||
call: func(p Provider) error { _, err := p.Create(context.Background(), CreateRequest{}); return err },
|
||||
wantOp: "create",
|
||||
wantResult: "quota_exceeded",
|
||||
},
|
||||
{
|
||||
name: "delete permanent",
|
||||
inner: &staticProvider{deleteErr: Wrap(ErrPermanent, "delete", "x", "id-1", nil)},
|
||||
call: func(p Provider) error { return p.Delete(context.Background(), "id-1") },
|
||||
wantOp: "delete",
|
||||
wantResult: "permanent",
|
||||
},
|
||||
{
|
||||
name: "unclassified list error is transient",
|
||||
inner: &staticProvider{listErr: errors.New("connection reset")},
|
||||
call: func(p Provider) error { _, err := p.ListByTag(context.Background()); return err },
|
||||
wantOp: "list",
|
||||
wantResult: "transient",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
rec := &fakeRecorder{}
|
||||
p := WithMetrics("gcp-eu", tc.inner, rec)
|
||||
|
||||
callErr := tc.call(p)
|
||||
|
||||
if len(rec.calls) != 1 {
|
||||
t.Fatalf("recorded %d calls, want 1", len(rec.calls))
|
||||
}
|
||||
want := recordedCall{provider: "gcp-eu", op: tc.wantOp, result: tc.wantResult}
|
||||
if rec.calls[0] != want {
|
||||
t.Errorf("recorded %+v, want %+v", rec.calls[0], want)
|
||||
}
|
||||
// The decorator must be transparent: errors pass through.
|
||||
if (tc.wantResult == "ok") != (callErr == nil) {
|
||||
t.Errorf("error passthrough broken: result %s but err %v", tc.wantResult, callErr)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
33
internal/provider/name.go
Normal file
33
internal/provider/name.go
Normal file
@@ -0,0 +1,33 @@
|
||||
package provider
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/base32"
|
||||
"strings"
|
||||
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
)
|
||||
|
||||
const namePrefix = "proxy-"
|
||||
|
||||
// NameFromUID derives a deterministic instance name from a Proxy CR's UID.
|
||||
// Provider.Create implementations key idempotency on this name: a repeat
|
||||
// call after a crash must find the existing instance by this name rather
|
||||
// than create a duplicate, and the reconciler relies on that to recover
|
||||
// providerID after a crash even if status was wiped.
|
||||
//
|
||||
// SHA-256 of the UID, truncated to the first 10 bytes, RFC 4648 base32
|
||||
// (no padding), lowercased: exactly 16 characters, for 22 total with the
|
||||
// "proxy-" prefix. That gives 80 bits of collision resistance — a birthday
|
||||
// collision only becomes likely around 2^40 objects, against a fleet of
|
||||
// tens — while satisfying GCP's instance name rules
|
||||
// (^[a-z]([-a-z0-9]{0,61}[a-z0-9])?$, <=63 chars) with headroom, since GCP
|
||||
// auto-names some resources (e.g. the boot disk) after the instance name.
|
||||
// base32's alphabet (A-Z2-7) is entirely legal once lowercased; base64 is
|
||||
// not (has '+', '/', and uppercase), and hex would need 20 characters for
|
||||
// the same 80 bits.
|
||||
func NameFromUID(uid types.UID) string {
|
||||
sum := sha256.Sum256([]byte(uid))
|
||||
enc := base32.StdEncoding.WithPadding(base32.NoPadding)
|
||||
return namePrefix + strings.ToLower(enc.EncodeToString(sum[:10]))
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user