Compare commits
10 Commits
a4a483acbc
...
801a9fbe5f
| Author | SHA1 | Date | |
|---|---|---|---|
| 801a9fbe5f | |||
| 71c00c40d1 | |||
| 1125f74221 | |||
| 5c408cc284 | |||
| 4282d73c74 | |||
| 05d490c0b8 | |||
| 7700358764 | |||
| ff859ebd84 | |||
| 4529594fb7 | |||
| ef1387dc01 |
@@ -44,13 +44,34 @@
|
||||
"Bash(cd /Users/jan.novak/srv/go/egress-proxies-operator *)",
|
||||
"Bash(echo \"build: $?\")",
|
||||
"Bash(echo \"vet: $?\")",
|
||||
"Bash(perl -i -pe 's{^\\\\t\\\\t\\\\t\\\\t\\\\t// TODO\\\\\\(user\\\\\\): Specify other spec details if needed\\\\.\\\\n}{\\\\t\\\\t\\\\t\\\\t\\\\t// A minimal, schema-valid spec so this placeholder test survives the\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t// CEL/CRD validation added in Step 1. Rewritten wholesale in Step 4\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t// alongside the real reconciler and envtest suite.\\\\n\\\\t\\\\t\\\\t\\\\t\\\\tSpec: crawlv1alpha1.ProxySpec{\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t\\\\tMode: crawlv1alpha1.ModeExternal,\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t\\\\tEndpoint: &crawlv1alpha1.EndpointSpec{Host: \"10.0.0.1\"},\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t},\\\\n}' internal/controller/proxy_controller_test.go)"
|
||||
"Bash(perl -i -pe 's{^\\\\t\\\\t\\\\t\\\\t\\\\t// TODO\\\\\\(user\\\\\\): Specify other spec details if needed\\\\.\\\\n}{\\\\t\\\\t\\\\t\\\\t\\\\t// A minimal, schema-valid spec so this placeholder test survives the\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t// CEL/CRD validation added in Step 1. Rewritten wholesale in Step 4\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t// alongside the real reconciler and envtest suite.\\\\n\\\\t\\\\t\\\\t\\\\t\\\\tSpec: crawlv1alpha1.ProxySpec{\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t\\\\tMode: crawlv1alpha1.ModeExternal,\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t\\\\tEndpoint: &crawlv1alpha1.EndpointSpec{Host: \"10.0.0.1\"},\\\\n\\\\t\\\\t\\\\t\\\\t\\\\t},\\\\n}' internal/controller/proxy_controller_test.go)",
|
||||
"Bash(go tool *)",
|
||||
"Bash(docker version *)",
|
||||
"Bash(curl -sI --max-time 5 https://hub.docker.com)",
|
||||
"Bash(kind get *)",
|
||||
"Bash(curl -s --max-time 5 \"https://hub.docker.com/v2/repositories/ubuntu/squid/tags?page_size=10\")",
|
||||
"Bash(python3 -c ' *)",
|
||||
"Bash(curl -s --max-time 5 \"https://hub.docker.com/v2/repositories/ubuntu/squid/tags?page_size=100\")",
|
||||
"Bash(go doc *)",
|
||||
"Bash(go list *)",
|
||||
"Bash(gofmt -w internal/provider/config.go)",
|
||||
"Bash(gofmt -l .)",
|
||||
"Bash(git restore *)",
|
||||
"Bash(make manifests *)",
|
||||
"Bash(make test *)",
|
||||
"Bash(KUBEBUILDER_ASSETS=\"/Users/jan.novak/srv/go/egress-proxies-operator/bin/k8s/1.36.2-darwin-arm64\" go test -race ./internal/controller/)",
|
||||
"Bash(grep -n 'func Channel' -A8 __CMDSUB_OUTPUT__/sigs.k8s.io/controller-runtime@v0.24.1/pkg/source/source.go)",
|
||||
"Bash(grep -n 'type GenericEvent' __CMDSUB_OUTPUT__/sigs.k8s.io/controller-runtime@v0.24.1/pkg/event/event.go)",
|
||||
"Bash(KUBEBUILDER_ASSETS=__TRACKED_VAR__/bin/k8s/1.36.2-darwin-arm64 go test -race ./...)",
|
||||
"Bash(cat >> *)"
|
||||
],
|
||||
"additionalDirectories": [
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/.claude",
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/docs/plans",
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/docs",
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/docs/prompts"
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/docs/prompts",
|
||||
"/tmp",
|
||||
"/Users/jan.novak/srv/go/egress-proxies-operator/docs/plans-executions"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
33
cmd/main.go
33
cmd/main.go
@@ -33,7 +33,6 @@ import (
|
||||
"sigs.k8s.io/controller-runtime/pkg/log/zap"
|
||||
"sigs.k8s.io/controller-runtime/pkg/metrics/filters"
|
||||
metricsserver "sigs.k8s.io/controller-runtime/pkg/metrics/server"
|
||||
"sigs.k8s.io/controller-runtime/pkg/webhook"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/controller"
|
||||
@@ -56,7 +55,6 @@ func init() {
|
||||
func main() {
|
||||
var metricsAddr string
|
||||
var metricsCertPath, metricsCertName, metricsCertKey string
|
||||
var webhookCertPath, webhookCertName, webhookCertKey string
|
||||
var enableLeaderElection bool
|
||||
var probeAddr string
|
||||
var secureMetrics bool
|
||||
@@ -70,15 +68,12 @@ func main() {
|
||||
"Enabling this will ensure there is only one active controller manager.")
|
||||
flag.BoolVar(&secureMetrics, "metrics-secure", true,
|
||||
"If set, the metrics endpoint is served securely via HTTPS. Use --metrics-secure=false to use HTTP instead.")
|
||||
flag.StringVar(&webhookCertPath, "webhook-cert-path", "", "The directory that contains the webhook certificate.")
|
||||
flag.StringVar(&webhookCertName, "webhook-cert-name", "tls.crt", "The name of the webhook certificate file.")
|
||||
flag.StringVar(&webhookCertKey, "webhook-cert-key", "tls.key", "The name of the webhook key file.")
|
||||
flag.StringVar(&metricsCertPath, "metrics-cert-path", "",
|
||||
"The directory that contains the metrics server certificate.")
|
||||
flag.StringVar(&metricsCertName, "metrics-cert-name", "tls.crt", "The name of the metrics server certificate file.")
|
||||
flag.StringVar(&metricsCertKey, "metrics-cert-key", "tls.key", "The name of the metrics server key file.")
|
||||
flag.BoolVar(&enableHTTP2, "enable-http2", false,
|
||||
"If set, HTTP/2 will be enabled for the metrics and webhook servers")
|
||||
"If set, HTTP/2 will be enabled for the metrics server")
|
||||
opts := zap.Options{
|
||||
Development: true,
|
||||
}
|
||||
@@ -102,23 +97,6 @@ func main() {
|
||||
tlsOpts = append(tlsOpts, disableHTTP2)
|
||||
}
|
||||
|
||||
// Initial webhook TLS options
|
||||
webhookTLSOpts := tlsOpts
|
||||
webhookServerOptions := webhook.Options{
|
||||
TLSOpts: webhookTLSOpts,
|
||||
}
|
||||
|
||||
if len(webhookCertPath) > 0 {
|
||||
setupLog.Info("Initializing webhook certificate watcher using provided certificates",
|
||||
"webhook-cert-path", webhookCertPath, "webhook-cert-name", webhookCertName, "webhook-cert-key", webhookCertKey)
|
||||
|
||||
webhookServerOptions.CertDir = webhookCertPath
|
||||
webhookServerOptions.CertName = webhookCertName
|
||||
webhookServerOptions.KeyName = webhookCertKey
|
||||
}
|
||||
|
||||
webhookServer := webhook.NewServer(webhookServerOptions)
|
||||
|
||||
// Metrics endpoint is enabled in 'config/default/kustomization.yaml'. The Metrics options configure the server.
|
||||
// More info:
|
||||
// - https://pkg.go.dev/sigs.k8s.io/controller-runtime@v0.24.1/pkg/metrics/server
|
||||
@@ -139,12 +117,8 @@ func main() {
|
||||
|
||||
// If the certificate is not specified, controller-runtime will automatically
|
||||
// generate self-signed certificates for the metrics server. While convenient for development and testing,
|
||||
// this setup is not recommended for production.
|
||||
//
|
||||
// TODO(user): If you enable certManager, uncomment the following lines:
|
||||
// - [METRICS-WITH-CERTS] at config/default/kustomization.yaml to generate and use certificates
|
||||
// managed by cert-manager for the metrics server.
|
||||
// - [PROMETHEUS-WITH-CERTS] at config/prometheus/kustomization.yaml for TLS certification.
|
||||
// this setup is not recommended for production. This project doesn't use cert-manager (no admission
|
||||
// webhooks, no other consumer of managed certs) -- pass real certs via the flags below if needed.
|
||||
if len(metricsCertPath) > 0 {
|
||||
setupLog.Info("Initializing metrics certificate watcher using provided certificates",
|
||||
"metrics-cert-path", metricsCertPath, "metrics-cert-name", metricsCertName, "metrics-cert-key", metricsCertKey)
|
||||
@@ -157,7 +131,6 @@ func main() {
|
||||
mgr, err := ctrl.NewManager(ctrl.GetConfigOrDie(), ctrl.Options{
|
||||
Scheme: scheme,
|
||||
Metrics: metricsServerOptions,
|
||||
WebhookServer: webhookServer,
|
||||
HealthProbeBindAddress: probeAddr,
|
||||
LeaderElection: enableLeaderElection,
|
||||
LeaderElectionID: "b47711d1.example.com",
|
||||
|
||||
@@ -4,13 +4,3 @@
|
||||
resources:
|
||||
- bases/crawl.example.com_proxies.yaml
|
||||
# +kubebuilder:scaffold:crdkustomizeresource
|
||||
|
||||
patches:
|
||||
# [WEBHOOK] To enable webhook, uncomment all the sections with [WEBHOOK] prefix.
|
||||
# patches here are for enabling the conversion webhook for each CRD
|
||||
# +kubebuilder:scaffold:crdkustomizewebhookpatch
|
||||
|
||||
# [WEBHOOK] To enable webhook, uncomment the following section
|
||||
# the following config is for teaching kustomize how to do kustomization for CRDs.
|
||||
#configurations:
|
||||
#- kustomizeconfig.yaml
|
||||
|
||||
@@ -1,12 +0,0 @@
|
||||
# This file is for teaching kustomize how to substitute name and namespace reference in CRD
|
||||
nameReference:
|
||||
- kind: Service
|
||||
version: v1
|
||||
fieldSpecs:
|
||||
- kind: CustomResourceDefinition
|
||||
version: v1
|
||||
group: apiextensions.k8s.io
|
||||
path: spec/conversion/webhook/clientConfig/service/name
|
||||
|
||||
varReference:
|
||||
- path: metadata/annotations
|
||||
@@ -18,20 +18,10 @@ resources:
|
||||
- ../crd
|
||||
- ../rbac
|
||||
- ../manager
|
||||
# [WEBHOOK] To enable webhook, uncomment all the sections with [WEBHOOK] prefix including the one in
|
||||
# crd/kustomization.yaml
|
||||
#- ../webhook
|
||||
# [CERTMANAGER] To enable cert-manager, uncomment all sections with 'CERTMANAGER'. 'WEBHOOK' components are required.
|
||||
#- ../certmanager
|
||||
# [PROMETHEUS] To enable prometheus monitor, uncomment all sections with 'PROMETHEUS'.
|
||||
#- ../prometheus
|
||||
# [METRICS] Expose the controller manager metrics service.
|
||||
- metrics_service.yaml
|
||||
# [NETWORK POLICY] Protect the /metrics endpoint and Webhook Server with NetworkPolicy.
|
||||
# Only Pod(s) running a namespace labeled with 'metrics: enabled' will be able to gather the metrics.
|
||||
# Only CR(s) which requires webhooks and are applied on namespaces labeled with 'webhooks: enabled' will
|
||||
# be able to communicate with the Webhook Server.
|
||||
#- ../network-policy
|
||||
|
||||
# Uncomment the patches line if you enable Metrics
|
||||
patches:
|
||||
@@ -48,14 +38,9 @@ patches:
|
||||
# target:
|
||||
# kind: Deployment
|
||||
|
||||
# [WEBHOOK] To enable webhook, uncomment all the sections with [WEBHOOK] prefix including the one in
|
||||
# crd/kustomization.yaml
|
||||
#- path: manager_webhook_patch.yaml
|
||||
# target:
|
||||
# kind: Deployment
|
||||
|
||||
# [CERTMANAGER] To enable cert-manager, uncomment all sections with 'CERTMANAGER' prefix.
|
||||
# Uncomment the following replacements to add the cert-manager CA injection annotations
|
||||
# [METRICS-WITH-CERTS] Uncomment the following replacements together with the patch
|
||||
# above to wire the metrics Service name/namespace into the cert-manager Certificate
|
||||
# and the Prometheus ServiceMonitor TLS config.
|
||||
#replacements:
|
||||
# - source: # Uncomment the following block to enable certificates for metrics
|
||||
# kind: Service
|
||||
@@ -116,119 +101,3 @@ patches:
|
||||
# delimiter: '.'
|
||||
# index: 1
|
||||
# create: true
|
||||
|
||||
# - source: # Uncomment the following block if you have any webhook
|
||||
# kind: Service
|
||||
# version: v1
|
||||
# name: webhook-service
|
||||
# fieldPath: .metadata.name # Name of the service
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPaths:
|
||||
# - .spec.dnsNames.0
|
||||
# - .spec.dnsNames.1
|
||||
# options:
|
||||
# delimiter: '.'
|
||||
# index: 0
|
||||
# create: true
|
||||
# - source:
|
||||
# kind: Service
|
||||
# version: v1
|
||||
# name: webhook-service
|
||||
# fieldPath: .metadata.namespace # Namespace of the service
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPaths:
|
||||
# - .spec.dnsNames.0
|
||||
# - .spec.dnsNames.1
|
||||
# options:
|
||||
# delimiter: '.'
|
||||
# index: 1
|
||||
# create: true
|
||||
|
||||
# - source: # Uncomment the following block if you have a ValidatingWebhook (--programmatic-validation)
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert # This name should match the one in certificate.yaml
|
||||
# fieldPath: .metadata.namespace # Namespace of the certificate CR
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: ValidatingWebhookConfiguration
|
||||
# fieldPaths:
|
||||
# - .metadata.annotations.[cert-manager.io/inject-ca-from]
|
||||
# options:
|
||||
# delimiter: '/'
|
||||
# index: 0
|
||||
# create: true
|
||||
# - source:
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPath: .metadata.name
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: ValidatingWebhookConfiguration
|
||||
# fieldPaths:
|
||||
# - .metadata.annotations.[cert-manager.io/inject-ca-from]
|
||||
# options:
|
||||
# delimiter: '/'
|
||||
# index: 1
|
||||
# create: true
|
||||
|
||||
# - source: # Uncomment the following block if you have a DefaultingWebhook (--defaulting )
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPath: .metadata.namespace # Namespace of the certificate CR
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: MutatingWebhookConfiguration
|
||||
# fieldPaths:
|
||||
# - .metadata.annotations.[cert-manager.io/inject-ca-from]
|
||||
# options:
|
||||
# delimiter: '/'
|
||||
# index: 0
|
||||
# create: true
|
||||
# - source:
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPath: .metadata.name
|
||||
# targets:
|
||||
# - select:
|
||||
# kind: MutatingWebhookConfiguration
|
||||
# fieldPaths:
|
||||
# - .metadata.annotations.[cert-manager.io/inject-ca-from]
|
||||
# options:
|
||||
# delimiter: '/'
|
||||
# index: 1
|
||||
# create: true
|
||||
|
||||
# - source: # Uncomment the following block if you have a ConversionWebhook (--conversion)
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPath: .metadata.namespace # Namespace of the certificate CR
|
||||
# targets: # Do not remove or uncomment the following scaffold marker; required to generate code for target CRD.
|
||||
# +kubebuilder:scaffold:crdkustomizecainjectionns
|
||||
# - source:
|
||||
# kind: Certificate
|
||||
# group: cert-manager.io
|
||||
# version: v1
|
||||
# name: serving-cert
|
||||
# fieldPath: .metadata.name
|
||||
# targets: # Do not remove or uncomment the following scaffold marker; required to generate code for target CRD.
|
||||
# +kubebuilder:scaffold:crdkustomizecainjectionname
|
||||
|
||||
@@ -1,2 +1,8 @@
|
||||
resources:
|
||||
- manager.yaml
|
||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||
kind: Kustomization
|
||||
images:
|
||||
- name: controller
|
||||
newName: example.com/egress-proxies-operator
|
||||
newTag: v0.0.1
|
||||
|
||||
@@ -1,27 +0,0 @@
|
||||
# This NetworkPolicy allows ingress traffic
|
||||
# with Pods running on namespaces labeled with 'metrics: enabled'. Only Pods on those
|
||||
# namespaces are able to gather data from the metrics endpoint.
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: NetworkPolicy
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
app.kubernetes.io/managed-by: kustomize
|
||||
name: allow-metrics-traffic
|
||||
namespace: system
|
||||
spec:
|
||||
podSelector:
|
||||
matchLabels:
|
||||
control-plane: controller-manager
|
||||
app.kubernetes.io/name: egress-proxies-operator
|
||||
policyTypes:
|
||||
- Ingress
|
||||
ingress:
|
||||
# This allows ingress traffic from any namespace with the label metrics: enabled
|
||||
- from:
|
||||
- namespaceSelector:
|
||||
matchLabels:
|
||||
metrics: enabled # Only from namespaces with this label
|
||||
ports:
|
||||
- port: 8443
|
||||
protocol: TCP
|
||||
@@ -1,2 +0,0 @@
|
||||
resources:
|
||||
- allow-metrics-traffic.yaml
|
||||
@@ -4,6 +4,14 @@ kind: ClusterRole
|
||||
metadata:
|
||||
name: manager-role
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- secrets
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- watch
|
||||
- apiGroups:
|
||||
- crawl.example.com
|
||||
resources:
|
||||
|
||||
185
docs/architecture.md
Normal file
185
docs/architecture.md
Normal file
@@ -0,0 +1,185 @@
|
||||
# Architecture
|
||||
|
||||
> **Status:** the operator is built through Step 5 (health engine) of
|
||||
> [docs/plans/2026-08-07-1747-proxy-operator.md](plans/2026-08-07-1747-proxy-operator.md).
|
||||
> This document currently covers the event/reconcile flow; the components
|
||||
> table and the Decisions section arrive with Step 10, and the diagrams
|
||||
> below grow as the lease store, discovery API, and orphan GC land.
|
||||
|
||||
## Event flow: cluster events → reconciler functions
|
||||
|
||||
Which functions run in response to which Kubernetes cluster events, as
|
||||
wired in `internal/controller/proxy_controller.go`.
|
||||
|
||||
### 1. How cluster events reach the reconciler
|
||||
|
||||
```text
|
||||
KUBERNETES CLUSTER EVENTS (wiring: SetupWithManager,
|
||||
───────────────────────── proxy_controller.go)
|
||||
|
||||
Proxy CR created / spec edited / Secret created / health transition
|
||||
status patched / delete requested updated / deleted (engine, see §6)
|
||||
│ │ │
|
||||
watch: For(&crawlv1alpha1.Proxy{}) watch: Watches( WatchesRawSource(
|
||||
│ &corev1.Secret{}, ...) source.Channel(
|
||||
│ │ r.HealthEvents, ...))
|
||||
│ r.proxiesForSecret(ctx, secret) │
|
||||
│ │ r.List(Proxies in namespace) │
|
||||
│ │ keeps those whose │
|
||||
│ │ spec.cloudInit.secretRef matches │
|
||||
│ ▼ │
|
||||
│ [reconcile.Request per matching Proxy] │
|
||||
▼ │ │
|
||||
┌────────────────────────────────────┴──────────────────────────┴──┐
|
||||
│ controller-runtime workqueue │◄── RequeueAfter
|
||||
│ (dedup by namespace/name, rate-limited, │ timers
|
||||
│ MaxConcurrentReconciles: 3) │◄── error backoff
|
||||
└──────────────────────────┬────────────────────────────────────────┘
|
||||
▼
|
||||
ProxyReconciler.Reconcile(ctx, req)
|
||||
```
|
||||
|
||||
The Secret watch makes *rotating a cloud-init Secret* a first-class event:
|
||||
it re-enqueues every Proxy referencing that Secret, which is how secret
|
||||
rotation triggers VM replacement even though the Proxy spec is untouched.
|
||||
|
||||
### 2. Inside `Reconcile` — dispatch and the single status write
|
||||
|
||||
```text
|
||||
Reconcile(ctx, req)
|
||||
│ r.Get(ctx, req.NamespacedName, &p) ── fetch the Proxy (NotFound → done)
|
||||
│ base := p.DeepCopy() ── snapshot for the diff
|
||||
│ defer patchStatusIfChanged(ctx, base, &p) ──────────────────────────┐
|
||||
│ │
|
||||
├─ p.DeletionTimestamp set ──► reconcileDelete(ctx, &p) │
|
||||
├─ p.Spec.Mode == External ──► reconcileExternal(ctx, &p) │
|
||||
└─ otherwise (Managed) ──────► reconcileManaged(ctx, &p) │
|
||||
▼
|
||||
patchStatusIfChanged (status.go)
|
||||
│ p.Status.ObservedGeneration = p.Generation
|
||||
│ p.Status.Phase = computePhase(&p)
|
||||
│ equality.Semantic.DeepEqual(base, p)?
|
||||
└─ changed → r.Status().Patch(...) ◄── the ONLY
|
||||
unchanged → no API call status write
|
||||
```
|
||||
|
||||
### 3. `reconcileManaged` — the state machine
|
||||
|
||||
```text
|
||||
reconcileManaged(ctx, p)
|
||||
│
|
||||
├─ controllerutil.AddFinalizer? ──► r.Update ──► return {} (watch event re-triggers)
|
||||
├─ permanent-failure latch (FindStatusCondition == PermanentError
|
||||
│ at this generation) ──► return {} (silent until spec edit)
|
||||
├─ r.Providers[p.Spec.Provider] missing ──► setProvisioned(PermanentError) → Failed
|
||||
│
|
||||
├─ resolveCloudInit(ctx, p) ──► r.Get(Secret) if secretRef (error → CloudInitError + backoff)
|
||||
├─ hash := specHash(p, cloudInit) (spechash.go)
|
||||
│
|
||||
├─ status.providerID == "" ─────────► prov.Create(CreateRequest{Name: NameFromUID(p.UID), ...})
|
||||
│ │ setSpecHash → r.Update (annotation)
|
||||
│ └ stage providerID + Provisioned=False/Provisioning
|
||||
│ ──► RequeueAfter: ProvisioningPoll
|
||||
│
|
||||
├─ annotation != hash, annotation == "" ──► adopt: setSpecHash → r.Update
|
||||
│ ──► RequeueAfter: RequeueNow
|
||||
├─ annotation != hash, annotation != "" ──► replaceInstance:
|
||||
│ prov.Get ─ NotFound → setSpecHash, clear ID/IP
|
||||
│ │ ──► RequeueNow (next pass creates)
|
||||
│ └ exists → prov.Delete, Provisioned=False/Replacing
|
||||
│ ──► RequeueAfter: DeletionPoll
|
||||
│
|
||||
└─ annotation == hash ──► prov.Get(providerID)
|
||||
├─ ErrNotFound ──► clear ID/IP ──► RequeueNow (next pass creates)
|
||||
├─ Provisioning ──► Provisioned=False ──► ProvisioningPoll
|
||||
├─ Running ──► status.ip = inst.IP,
|
||||
│ Provisioned=True/Created,
|
||||
│ applyHealth (see §6) ──► DriftPoll
|
||||
└─ Stopped/Termin. ──► prov.Delete (cattle) ──► DeletionPoll
|
||||
|
||||
any provider error ──► providerFailure(p, err) ── provider.Class(err):
|
||||
├─ ErrQuotaExceeded ──► condition QuotaExceeded ──► RequeueAfter: QuotaRetry (nil error)
|
||||
├─ ErrPermanent ──► condition PermanentError ──► phase Failed, no retry
|
||||
└─ ErrTransient ──► return err ──► workqueue exponential backoff
|
||||
```
|
||||
|
||||
### 4. `reconcileDelete` and `reconcileExternal`
|
||||
|
||||
```text
|
||||
reconcileDelete(ctx, p) reconcileExternal(ctx, p)
|
||||
├─ no finalizer ──► return {} │ status.ip = spec.endpoint.host
|
||||
├─ providerID == "" ──► RemoveFinalizer │ setProvisioned(True/ExternalEndpoint)
|
||||
│ → r.Update → object actually deleted │ applyHealth (see §6)
|
||||
├─ prov.Get → ErrNotFound ──► RemoveFinalizer └─ return {} (no finalizer,
|
||||
│ → r.Update → object actually deleted no provider calls ever)
|
||||
└─ exists ──► prov.Delete
|
||||
→ Provisioned=False/Deleting
|
||||
──► RequeueAfter: DeletionPoll (poll until gone)
|
||||
```
|
||||
|
||||
### 5. What provider calls do back in the cluster (kubernetes pod provider)
|
||||
|
||||
```text
|
||||
prov.Create ──► buildPod (pure) ──► client.Create(corev1.Pod) ─┐ these cause Pod events,
|
||||
prov.Get ──► client.Get(Pod) → phase/IP → InstanceState │ but the operator does NOT
|
||||
prov.Delete ──► client.Delete(Pod, tolerate NotFound) │ watch Pods — it observes
|
||||
prov.ListByTag ─► client.List(Pods by labels, all namespaces) ─┘ them by polling prov.Get
|
||||
on each RequeueAfter tick
|
||||
```
|
||||
|
||||
The reconciler never watches provider-side resources (Pods now, GCP VMs
|
||||
later). All instance-state observation is poll-based through the
|
||||
`Provider` interface, so the same flow works identically for a cloud API
|
||||
that has no watch mechanism at all.
|
||||
|
||||
### 6. Health engine (`internal/health/`) — probes and transitions
|
||||
|
||||
The engine is a leader-elected manager Runnable with its own goroutines,
|
||||
independent of the workqueue. It owns health *state*; the reconciler owns
|
||||
its *representation* in status — that split keeps exactly one writer of
|
||||
`.status` and makes write-only-on-transition fall out for free.
|
||||
|
||||
```text
|
||||
Engine.Start(ctx) (engine.go)
|
||||
├─ spawns Workers (8) probe goroutines ◄─┐
|
||||
└─ ticker loop (Tick = 1s): │ jobs channel (non-blocking send;
|
||||
tick(ctx, now, jobs) │ saturated pool → retry next tick)
|
||||
│ Reader.List(Proxies) ── from the manager cache
|
||||
│ per proxy: skip if no IP/host or deleting (state pruned →
|
||||
│ a replaced instance starts with fresh counters)
|
||||
│ newState: seed verdict from an existing Healthy condition
|
||||
│ (leader handover), jitter first probe across the interval
|
||||
│ due && !inFlight ──► jobs ◄── probe worker picks up
|
||||
└ prune states for proxies gone from the cache
|
||||
│
|
||||
probe(ctx, proxyURL, hc, tls) (probe.go)
|
||||
│ fresh transport per probe, DisableKeepAlives=true
|
||||
│ (load-bearing: keep-alives would cache the CONNECT
|
||||
│ tunnel and later probes would never re-exercise it)
|
||||
│ https probe URL ⇒ CONNECT through the proxy + TLS inside
|
||||
└ success = err == nil AND expected status code
|
||||
│
|
||||
record(job, result, now) ── under one mutex
|
||||
│ counters: consecOK/consecFail; verdict flips only at
|
||||
│ successThreshold / failureThreshold
|
||||
│ emit ONLY on: first-ever verdict │ threshold flip │
|
||||
│ latency Δ > max(20ms, 50% of reported) rate-limited
|
||||
│ to one report per MinReportInterval (60s)
|
||||
▼
|
||||
Events chan (buffered 64, non-blocking send;
|
||||
on drop the reported markers do NOT advance → next probe retries)
|
||||
│
|
||||
▼
|
||||
source.Channel → workqueue → Reconcile (see §1)
|
||||
│
|
||||
▼
|
||||
r.applyHealth(p) ── reads Engine.Snapshot(key) (status.go)
|
||||
stages the Healthy condition + latencyMillis +
|
||||
lastHealthCheckTime; computePhase turns Provisioned=True
|
||||
+ Healthy=True/False into phase Ready / Unhealthy
|
||||
```
|
||||
|
||||
Consequence worth knowing: `status.lastHealthCheckTime` is the time of the
|
||||
last *status-affecting* probe, not the most recent probe — suppressed
|
||||
probes deliberately never write status. True probe recency will live in
|
||||
metrics (Step 9).
|
||||
@@ -7,9 +7,9 @@ Pairs with [docs/plans/2026-08-07-1747-proxy-operator.md](../plans/2026-08-07-17
|
||||
- [x] Step 0 — Branch and scaffold
|
||||
- [x] Step 1 — API types (`api/v1alpha1/proxy_types.go`)
|
||||
- [x] Step 2 — Provider contract (`internal/provider/`)
|
||||
- [ ] Step 3 — Mock provider (`internal/provider/mock/`)
|
||||
- [ ] Step 4 — Reconciler (`internal/controller/`)
|
||||
- [ ] Step 5 — Health engine (`internal/health/`)
|
||||
- [x] Step 3 — Kubernetes pod provider (`internal/provider/kubernetes/`; first built as an in-memory mock, then replaced — see the two Step 3 sections below)
|
||||
- [x] Step 4 — Reconciler (`internal/controller/`)
|
||||
- [x] Step 5 — Health engine (`internal/health/`)
|
||||
- [ ] Step 6 — Lease store (`internal/lease/`)
|
||||
- [ ] Step 7 — Discovery API (`internal/discovery/`)
|
||||
- [ ] Step 8 — GCP provider (`internal/provider/gcp/`)
|
||||
@@ -199,3 +199,498 @@ built in (rejects unknown fields, which is what "fail fast on unknown type"
|
||||
in the plan actually needs) and was already pulled in transitively by the
|
||||
k8s.io toolchain, so no new dependency was added — `go mod tidy` just
|
||||
promoted it from indirect to direct.
|
||||
|
||||
## Step 3 — Mock provider (`internal/provider/mock/`)
|
||||
|
||||
Started from the plan's design (state as a pure function of an injectable
|
||||
clock, no background timers) but had to redesign the "real proxy" part
|
||||
before writing any code, once a load-bearing assumption turned out false.
|
||||
|
||||
The plan (and the earlier decision to make the mock run a real proxy)
|
||||
assumed each instance could get its own loopback address —
|
||||
`127.0.0.1:<port>` per instance, "a fake IP from a private range." Verified
|
||||
that assumption directly before committing to it:
|
||||
|
||||
```bash
|
||||
cat <<'EOF' > /tmp/loopbacktest.go
|
||||
package main
|
||||
import ("fmt"; "net")
|
||||
func main() {
|
||||
for _, addr := range []string{"127.0.0.2:0", "127.0.0.55:0", "127.1.2.3:0"} {
|
||||
l, err := net.Listen("tcp", addr)
|
||||
if err != nil { fmt.Printf("%s: FAIL: %v\n", addr, err); continue }
|
||||
fmt.Printf("%s: OK\n", addr)
|
||||
l.Close()
|
||||
}
|
||||
}
|
||||
EOF
|
||||
go run /tmp/loopbacktest.go
|
||||
# 127.0.0.2:0: FAIL: listen tcp 127.0.0.2:0: bind: can't assign requested address
|
||||
# 127.0.0.55:0: FAIL: listen tcp 127.0.0.55:0: bind: can't assign requested address
|
||||
# 127.1.2.3:0: FAIL: listen tcp 127.1.2.3:0: bind: can't assign requested address
|
||||
```
|
||||
|
||||
Only `127.0.0.1` binds on macOS without `sudo ifconfig lo0 alias ... up` —
|
||||
Linux routes the whole `127.0.0.0/8` block to loopback by default, macOS
|
||||
doesn't. That's not something the operator can or should do at runtime, so
|
||||
per-instance loopback IPs were out. There's also a second problem the
|
||||
per-instance-IP design didn't solve anyway: `EffectivePort()`
|
||||
(`api/v1alpha1/helpers.go`, Step 1) is computed purely from `spec.port`
|
||||
with no channel for a provider to report back a different *port* — so
|
||||
whatever a mock instance actually listens on has to be the literal port the
|
||||
reconciler will pass through `CreateRequest.Port`, not an OS-assigned
|
||||
ephemeral one.
|
||||
|
||||
Redesigned around one real listener **per port**, shared and
|
||||
reference-counted across every instance that uses it, rather than one
|
||||
listener per instance (`proxy.go`, `sharedProxies`). This fixes both
|
||||
problems at once: every instance binds the same `127.0.0.1` (no OS issue),
|
||||
and any number of instances can share a port without conflict since it's
|
||||
the exact same underlying listener. Deliberately made the refcounting
|
||||
**package-level**, not a field on `mock.Provider`, because a bound TCP port
|
||||
is a genuinely process-global OS resource — two separately configured
|
||||
mock-typed provider entries (e.g. two named `"mock"` instances in
|
||||
`providers-config.yaml`) would otherwise both try to bind the same default
|
||||
port and the second one would just fail. This is a case where a package
|
||||
global is the correct model, not a shortcut: it mirrors an OS-level
|
||||
singleton, not application state.
|
||||
|
||||
Instance lifecycle otherwise follows the plan exactly: `Get`/`ListByTag`
|
||||
derive `Provisioning → Running → Terminated → purged (ErrNotFound)` from
|
||||
`createdAt`/`deletedAt` compared against an injectable clock, with the real
|
||||
proxy listener acquired lazily on the first observed `Running` and released
|
||||
on `Delete` (or lazily on purge, so orphaned records can't leak a
|
||||
reference). `Create` is idempotent by name. Fault injection is wired both
|
||||
ways per the plan: `MockConfig.FailNextCreates`/`FailWith` for the demo
|
||||
config, `InjectCreateFailures(n, class)` for tests.
|
||||
|
||||
Tests initially had a real flake, caught by running with `-race -count=3`
|
||||
rather than trusting one green run:
|
||||
|
||||
```bash
|
||||
go test -race -v ./internal/provider/mock/... 2>&1 | tail -5
|
||||
# --- FAIL: TestAcquireProxy_sharedAcrossAcquires
|
||||
# proxy_test.go:37: port not released after last reference: bind: address already in use
|
||||
```
|
||||
|
||||
Root cause wasn't the refcounting logic — it was the test helper. `freePort`
|
||||
asked the OS for a free port by binding to `:0` and immediately closing it,
|
||||
which is a classic TOCTOU race under `t.Parallel()`: two tests can be handed
|
||||
the same "free" port before either actually claims it, since nothing holds
|
||||
it open in between. Fixed by replacing the OS-asks approach with a
|
||||
monotonic counter (`20000 + atomic.Int32`) — these tests only need a port
|
||||
unique *within this test run*, not one verified free by the OS at an
|
||||
instant in time, so guaranteeing uniqueness outright is both simpler and
|
||||
correct where the "ask and hope" approach wasn't. Reran `-race -count=3`
|
||||
clean afterward.
|
||||
|
||||
Added tests beyond the plan's list to close real coverage gaps rather than
|
||||
stopping at "green": config-override branches in `New`, all four
|
||||
`failClassFromString` branches, `ListByTag`'s purge-on-list and
|
||||
Running/IP-inclusion paths, and a plain-`http://` forwarding test
|
||||
(`handleForward`) alongside the CONNECT one, since a `probeURL` override
|
||||
could use either scheme. Landed at 91.1% coverage; the remainder is
|
||||
OS-failure branches (bind errors, hijack failures) not worth simulating for
|
||||
a prototype.
|
||||
|
||||
The `TestProvider_realProxyTunnelsConnect` test is the one that actually
|
||||
matters most here: it opens real sockets end to end — mock `Create` →
|
||||
`Get` past `provisionDelay` → real `http.Client` with
|
||||
`Transport.Proxy` dialing through the mock's CONNECT tunnel to a real
|
||||
`httptest.NewTLSServer` — and gets a real `204` back. That's the concrete
|
||||
proof the "mock runs a real proxy" decision actually delivers a genuine
|
||||
end-to-end healthcheck, not a simulated one.
|
||||
|
||||
`internal/provider/mock` at 91.1% coverage. `make test` green across the
|
||||
whole repo.
|
||||
|
||||
## Step 3 (revised) — Kubernetes pod provider replaces the mock
|
||||
|
||||
After the mock provider above was built and working, the user pushed back:
|
||||
it felt too far from the real system to build confidence in, and they'd
|
||||
rather have simpler, more "real" code than a fast-but-simulated test
|
||||
double — a `kind`-based verification pass "once in a while" is an
|
||||
acceptable trade. They proposed replacing it outright with a provider that
|
||||
creates real Pods in the operator's own cluster, rather than keeping mock
|
||||
around as a fallback.
|
||||
|
||||
**Talked through the trade-off before agreeing to it, since it wasn't free
|
||||
of downsides.** Pods sharing a cluster's egress IPs don't solve what this
|
||||
operator actually exists for (routing around IP-based rate limiting needs
|
||||
genuinely distinct egress paths — that's still only `gcp`), and `envtest`
|
||||
has no kubelet, so a Pod-based provider can never be exercised by the fast
|
||||
test suite either. Net effect: this is a replacement for the mock's role
|
||||
in local dev/CI confidence-building, not a new alternative to GCP, and the
|
||||
reconciler's own state-machine tests (Step 4) still need a minimal
|
||||
in-test stub `Provider` — a handful of lines in the test file, not a
|
||||
package with its own config format or fault-injection surface, which is
|
||||
the specific kind of complexity the user was pushing back on.
|
||||
|
||||
**Picked the proxy software by actually checking what's out there,** not
|
||||
by guessing a Docker Hub path:
|
||||
|
||||
```bash
|
||||
curl -s "https://hub.docker.com/v2/repositories/vimagick/tinyproxy/" | head -c 400
|
||||
# last_updated 2021-07-22 — stale
|
||||
curl -s "https://hub.docker.com/v2/repositories/ubuntu/squid/" | head -c 400
|
||||
# last_updated 2026-08-07T04:36:14Z — updated the same day, Canonical-published, 50M+ pulls
|
||||
```
|
||||
|
||||
`ubuntu/squid` won clearly: actively maintained (rebuilt the same day this
|
||||
check ran), official publisher, and — since it's a public image — `kind`
|
||||
nodes pull it directly, no build/load step needed for the quickstart.
|
||||
Pinned to `6.6-24.04_edge` (Ubuntu 24.04 LTS base) rather than floating
|
||||
`latest`, for reproducibility.
|
||||
|
||||
**Design, in `internal/provider/kubernetes/`:**
|
||||
|
||||
- **`pod.go`** — `buildPod` is a pure function (mirrors the GCP provider's
|
||||
planned `buildInsertRequest`): builds a `corev1.Pod` with one Squid
|
||||
container. Config is generated in Go and passed via a `SQUID_CONF` env
|
||||
var that the container's command writes to `/etc/squid/squid.conf`
|
||||
before `exec squid` — deliberately not a separate `ConfigMap`, so
|
||||
there's still exactly one Kubernetes object per proxy instance to
|
||||
create, track, and clean up. The config itself is intentionally
|
||||
permissive (`http_access allow all`, `via off`, `forwarded_for off`) —
|
||||
documented in-code as a prototype-for-a-private-cluster choice, the same
|
||||
posture the spec already takes toward cloud-init on the GCP provider
|
||||
(installing/configuring proxy software is explicitly out of scope
|
||||
there; this provider doesn't try to do more).
|
||||
- **`kubernetes.go`** — `Create`/`Get`/`Delete`/`ListByTag` against a
|
||||
`client.Client`. **`providerID` is `<namespace>/<podName>`**, parsed
|
||||
with `k8s.io/client-go/tools/cache.SplitMetaNamespaceKey` — the exact
|
||||
same reasoning as the GCP provider's planned zone-qualified providerID
|
||||
(Step 8): `Get`/`Delete` need to be self-contained without re-deriving
|
||||
where the resource lives. State mapping collapses `Succeeded`/`Failed`/
|
||||
`Unknown` all into `Terminated`, since the reconciler already treats
|
||||
Stopped and Terminated identically (delete + recreate) — no finer
|
||||
distinction would change any behavior. `Running` with no `PodIP` yet
|
||||
maps to `Provisioning`, not `Running`, so an empty IP is never
|
||||
published — the same rule the plan already called out for GCP's
|
||||
`RUNNING`-without-`NatIP` case.
|
||||
- **Client construction is the one genuinely new pattern this provider
|
||||
needed** that GCP/mock didn't: it builds its own `client.Client` via
|
||||
`ctrl.GetConfig()`, which auto-detects in-cluster config when running as
|
||||
a Pod and falls back to the local kubeconfig otherwise. That's what
|
||||
makes `make run` against a local `kind` cluster and running in-cluster
|
||||
use the exact same code path with zero provider-specific wiring in
|
||||
`cmd/main.go`. The corresponding risk: `New()` must never run under
|
||||
`go test` — it would happily connect to whatever real cluster the
|
||||
developer's kubeconfig points at. Solved the same way `mock.Provider`
|
||||
solved clock injection: an unexported `newWithClient(c, cfg)` constructor
|
||||
that tests call directly, bypassing `ctrl.GetConfig()` entirely.
|
||||
`New()` sits at 0% test coverage, deliberately — it's the one function
|
||||
that must stay untested by design.
|
||||
- **RBAC implication worth flagging now** (implemented in Step 10):
|
||||
`ListByTag` lists Pods across every namespace, not just the namespace(s)
|
||||
Proxies live in, because orphan GC needs to find every Pod this operator
|
||||
tagged regardless of where it landed. That means the operator's Pod
|
||||
permissions have to be a `ClusterRole`, not scoped to a single
|
||||
namespace — a real trade-off against the "fleet lives in one namespace,
|
||||
keeps RBAC simple" principle the spec states for the CRD itself.
|
||||
Documented rather than special-cased away.
|
||||
- **One known simplification, documented in a code comment rather than
|
||||
solved:** Kubernetes surfaces both RBAC-denied and quota-exceeded as the
|
||||
same 403 Forbidden, and `apierrors` has no helper to tell them apart.
|
||||
Both classify as `ErrPermanent` — the safer of the two defaults (stop
|
||||
retrying rather than hammering an API server that will never allow the
|
||||
request), but a real ResourceQuota failure that would clear once other
|
||||
proxies are deleted won't get the 5-minute-backoff-and-retry treatment
|
||||
`ErrQuotaExceeded` gives on the GCP path.
|
||||
|
||||
**Testing:** unlike the mock, this package's tests use
|
||||
`sigs.k8s.io/controller-runtime/pkg/client/fake` — real `corev1.Pod`
|
||||
objects, the real `client.Client` interface, not a hand-rolled in-memory
|
||||
map. That's a strictly more realistic test double than what mock.Provider
|
||||
was, while still being fast (no cluster, no kubelet) — it just can't prove
|
||||
a container actually starts and serves traffic, which is exactly the gap
|
||||
the `kind`-based verification pass is for. 77.6% coverage; the only
|
||||
meaningfully uncovered function is `New()` itself, deliberately.
|
||||
|
||||
`make test` green across the whole repo (`go build`/`go vet` clean,
|
||||
`internal/provider` 96.0%, `internal/provider/kubernetes` 77.6%,
|
||||
`internal/provider/registry` 100%, unchanged).
|
||||
|
||||
## Cleanup — removed unused cert-manager/webhook scaffolding
|
||||
|
||||
Not a plan step; the user asked for this directly after I explained what
|
||||
`make test-e2e` currently does, and didn't want unused scaffold machinery
|
||||
carried forward into later work.
|
||||
|
||||
kubebuilder's generic scaffold assumes a project *might* grow admission
|
||||
webhooks later, so it wires up webhook TLS-cert machinery and an
|
||||
unconditional cert-manager install in the e2e suite defensively. Checked
|
||||
whether any of it was actually load-bearing before touching anything:
|
||||
|
||||
```bash
|
||||
grep -rn "cert-manager\|certmanager\|CertManager" config/
|
||||
# only inside commented-out [CERTMANAGER] blocks in kustomization.yaml —
|
||||
# the whole block is inert, never uncommented
|
||||
grep -n "webhook" PROJECT
|
||||
# no output — kubebuilder create webhook was never run
|
||||
```
|
||||
|
||||
Confirmed nothing here does anything for this project — no `config/webhook/`
|
||||
exists, no `+kubebuilder:webhook` markers exist anywhere, and the spec's
|
||||
own non-goals explicitly rule out admission webhooks and cert-manager
|
||||
wiring forever, not just "not yet."
|
||||
|
||||
Removed:
|
||||
|
||||
- **`cmd/main.go`**: the `webhook` import, the three `webhook-cert-*`
|
||||
flags, the `webhookServerOptions`/`webhookServer` construction, and the
|
||||
`WebhookServer:` field on `ctrl.Options` — the manager runs with no
|
||||
webhook server at all now, which is correct since nothing registers one.
|
||||
Left the metrics-cert flags alone (`--metrics-cert-path` etc.) — those
|
||||
let real certs be mounted for the metrics endpoint without cert-manager,
|
||||
which is unrelated to webhooks and still useful. Reworded a comment that
|
||||
said "TODO(user): If you enable certManager..." since that will never
|
||||
happen here.
|
||||
- **`test/e2e/e2e_suite_test.go`**: the unconditional cert-manager install
|
||||
in `BeforeSuite` and matching uninstall in `AfterSuite`
|
||||
(`setupCertManager`/`teardownCertManager`/`shouldCleanupCertManager`),
|
||||
and the doc comment claiming the suite "requires Kind and CertManager"
|
||||
(it only requires Kind now).
|
||||
- **`test/utils/utils.go`**: `InstallCertManager`, `UninstallCertManager`,
|
||||
`IsCertManagerCRDsInstalled`, and their now-unused `warnError` helper and
|
||||
`certmanagerVersion`/`certmanagerURLTmpl` constants. Also removed
|
||||
`UncommentCode` — grepped first and confirmed it had zero callers even
|
||||
before this cleanup; it was dead scaffold code from the start, unrelated
|
||||
to cert-manager, just found while in there.
|
||||
|
||||
Left the inert commented-out `[WEBHOOK]`/`[CERTMANAGER]` blocks in the
|
||||
`config/*/kustomization.yaml` files and the
|
||||
`+kubebuilder:scaffold:e2e-webhooks-checks`-style marker comments in
|
||||
`test/e2e/e2e_test.go` alone — those are standard kubebuilder codegen
|
||||
anchors and pure comments with no runtime behavior, unlike the cert-manager
|
||||
install this cleanup actually removed. Stripping every trace of "webhook"
|
||||
from every scaffold comment across the tree would be a much bigger, purely
|
||||
cosmetic diff for no behavioral benefit; this cleanup targeted the things
|
||||
that were actually *doing* something.
|
||||
|
||||
Verified with both build tags, since `test/e2e` only compiles under `-tags=e2e`:
|
||||
|
||||
```bash
|
||||
go build ./... && go vet ./...
|
||||
go build -tags=e2e ./... && go vet -tags=e2e ./...
|
||||
```
|
||||
|
||||
Both clean. `make test` green across the whole repo, unchanged from before
|
||||
the cleanup.
|
||||
|
||||
## Cleanup — lean-down audit of Steps 0–3
|
||||
|
||||
Planned and approved separately in
|
||||
[docs/plans/2026-08-08-1335-lean-scaffold-cleanup.md](../plans/2026-08-08-1335-lean-scaffold-cleanup.md).
|
||||
Prompted by the user asking (a) why webhook machinery existed at all when the
|
||||
spec's §12 says "do NOT build: admission webhooks, cert-manager wiring", and
|
||||
(b) for a full audit of completed work so the project starts as lean as
|
||||
possible.
|
||||
|
||||
**Why the webhook machinery existed — for the record.** `kubebuilder init`
|
||||
emits it unconditionally: the active webhook-server wiring in `cmd/main.go`,
|
||||
cert-manager install in the e2e utils, and commented kustomize anchor blocks
|
||||
all arrive with `init`, not with `create webhook` (never run here). At Step 0
|
||||
the scaffold was deliberately committed untouched as a reviewable baseline
|
||||
(only `.github/` stripped), and spec §3's "config/ (scaffold-generated, kept
|
||||
working)" was read as license to keep the rest. The process gap: §12's
|
||||
non-goals deserved a pruning pass immediately *after* the baseline commit,
|
||||
especially for the parts that actually did something. The user caught it, not
|
||||
the build process.
|
||||
|
||||
**Could `init` have skipped it?** No — verified against the v4.15.0 binary,
|
||||
not from memory:
|
||||
|
||||
```bash
|
||||
kubebuilder init --help
|
||||
# flags: --domain --repo --owner --license(-file) --multigroup --namespaced
|
||||
# --fetch-deps --skip-go-version-check --project-version --plugins
|
||||
# nothing subtracts features; optional plugins (helm, grafana, deploy-image,
|
||||
# autoupdate) are all additive — there is no "minimal" plugin
|
||||
```
|
||||
|
||||
Scaffold-then-prune is the only supported path to a lean baseline. Worth
|
||||
knowing at the *next* project bootstrap: plan the pruning pass as part of
|
||||
scaffolding, not as a later discovery. (`--namespaced` was the one
|
||||
arguably-applicable flag, but it conflicts with the kubernetes-pod provider's
|
||||
cluster-wide `ListByTag` for orphan GC — cluster-scoped was correct.)
|
||||
|
||||
**Audit result for the Go code: clean.** Nothing beyond spec that isn't a
|
||||
justified, already-logged deviation (`Instance.UID`/`CreatedAt`,
|
||||
`MaxLeases *int32`, `HealthCheck default={}`, registry-as-parameter,
|
||||
kubernetes provider per user decision). Trivial extras (`shortName=px`,
|
||||
`MaxProperties=32`) stay.
|
||||
|
||||
**Scope lines drawn by the user during plan review** — recorded because they
|
||||
shape what "lean" means for this repo going forward:
|
||||
|
||||
- Developer tooling is exempt: `.golangci.yml`, `.custom-gcl.yml`,
|
||||
`.devcontainer/`, `AGENTS.md`, Makefile `lint`/`docker-buildx`/
|
||||
`build-installer` targets all stay ("tooling around the project is cool, i
|
||||
just want the application code produced to start as lean as possible").
|
||||
- Monitoring manifests stay: all of `config/prometheus/`, plus the paired
|
||||
metrics-TLS-via-cert-manager plumbing
|
||||
(`config/default/cert_metrics_manager_patch.yaml` and the metrics-certs/
|
||||
ServiceMonitor halves of the commented replacements block) — removing half
|
||||
of a pair would leave dangling comment references.
|
||||
- All RBAC manifests stay, including the `proxy_admin/editor/viewer` helper
|
||||
ClusterRoles whose own headers say "not used by the project itself".
|
||||
- Network policies go: "we do not need any network policies at the moment".
|
||||
|
||||
**Removed:**
|
||||
|
||||
- `config/network-policy/` (kustomization.yaml, allow-metrics-traffic.yaml)
|
||||
plus its commented `#- ../network-policy` line in the default kustomization.
|
||||
- Webhook-only remnants in `config/default/kustomization.yaml`: the commented
|
||||
`#- ../webhook` / `#- ../certmanager` resource lines, the
|
||||
`manager_webhook_patch.yaml` patch reference, and the webhook halves of the
|
||||
replacements block (serving-cert → Validating/Mutating WebhookConfiguration
|
||||
cainjection, conversion webhook, and the
|
||||
`+kubebuilder:scaffold:crdkustomizecainjection*` markers — anchors only for
|
||||
`kubebuilder create webhook`, permanently a non-goal).
|
||||
- The two commented `[WEBHOOK]` blocks in `config/crd/kustomization.yaml`
|
||||
(conversion patches + the `configurations:` reference), along with the empty
|
||||
`patches:` key they lived under and the `crdkustomizewebhookpatch` marker.
|
||||
Kept the one-line `crdkustomizeresource` marker — `kubebuilder create api`
|
||||
could legitimately run again.
|
||||
- `config/crd/kustomizeconfig.yaml` — only consumer was the removed
|
||||
`configurations:` block.
|
||||
|
||||
Verified: `bin/kustomize build config/default` and `... config/crd` both
|
||||
render cleanly (no dangling references), `go build`/`go vet` clean with and
|
||||
without `-tags=e2e`, `make test` green with coverage numbers identical to
|
||||
pre-cleanup.
|
||||
|
||||
## Step 4 — Reconciler (`internal/controller/`)
|
||||
|
||||
Implemented the state machine per the plan's action table:
|
||||
`proxy_controller.go` (dispatch + managed/external/delete paths, cloud-init
|
||||
resolution, spec-hash annotation persistence, Secret→Proxy watch mapping),
|
||||
`status.go` (condition reasons, `computePhase`, the single deferred
|
||||
`patchStatusIfChanged`), and `spechash.go` (explicit
|
||||
`{placement, resolved cloud-init, port}` hash input, SHA-256 hex). Tests:
|
||||
the action-table suite against a fake client with an in-test `stubProvider`
|
||||
(`reconcile_test.go`), `computePhase` truth table, spec-hash
|
||||
stability/normalization/sensitivity tables, and a rewritten envtest suite
|
||||
(`proxy_controller_test.go`) driving full lifecycles — provision→Running,
|
||||
spec-change replacement, finalizer deletion, External tracking — against
|
||||
the real apiserver with real CRD defaulting.
|
||||
|
||||
**Deviation from the plan's `Requeue: true` rows:** `ctrl.Result{Requeue}`
|
||||
is deprecated in controller-runtime v0.24 (verified in the vendored source,
|
||||
`pkg/reconcile/reconcile.go`: "Deprecated: Use `RequeueAfter` instead"), and
|
||||
golangci's staticcheck would flag it. Those rows use a fifth configurable
|
||||
interval instead, `RequeueNow` (default 1s) — same "process the next state
|
||||
promptly" semantics, still shrinkable in tests like the other four.
|
||||
|
||||
**A real bug the new tests caught on their first run** (both the fake-client
|
||||
and envtest suites, independently): in the replacement path's
|
||||
instance-is-gone branch, the status clear (`ProviderID = ""`) was staged
|
||||
*before* `setSpecHash`'s metadata `Update` — and `client.Update` refreshes
|
||||
the whole object from the server's response, *including status*, so the
|
||||
staged clear was silently overwritten and the proxy wedged with a stale
|
||||
providerID. Fix: stage status changes only after any metadata Update
|
||||
(the create branch already did it in that order). Worth remembering for
|
||||
every future reconciler: **`r.Update` clobbers in-memory status staged
|
||||
before it.**
|
||||
|
||||
Two judgment calls the plan left open, now documented in code:
|
||||
|
||||
- `computePhase` maps Provisioned=True with no Healthy verdict yet to
|
||||
`Provisioning`, not `Ready` — a proxy nobody has probed shouldn't be
|
||||
advertised as Ready. Health (Step 5) flips it.
|
||||
- `deletionFailure` (the finalizer path's error handler) never latches
|
||||
`ErrPermanent` the way `providerFailure` does — latching there would
|
||||
wedge the object forever with no retry; it keeps retrying visibly
|
||||
instead.
|
||||
|
||||
The permanent-failure latch compares the condition's `observedGeneration`
|
||||
against the CR generation, so a spec edit automatically clears Failed and
|
||||
retries — no manual annotation-poking needed to recover.
|
||||
|
||||
Verification:
|
||||
|
||||
```bash
|
||||
make test # regenerates manifests (role.yaml gains secrets get;list;watch), envtest green
|
||||
KUBEBUILDER_ASSETS="$PWD/bin/k8s/1.36.2-darwin-arm64" go test -race ./internal/controller/
|
||||
go test -short ./internal/controller/ # 0.6s — envtest suite correctly skipped
|
||||
go build -tags=e2e ./... && go vet -tags=e2e ./...
|
||||
```
|
||||
|
||||
`internal/controller` at 75.9% coverage; the envtest suite runs in ~6s and
|
||||
is now guarded by `testing.Short()` per the testing conventions.
|
||||
|
||||
Worth noting: the envtest specs simulate instance state by mutating the
|
||||
stub between direct `Reconcile` calls rather than running the manager —
|
||||
deterministic and fast, at the cost of not exercising watch-driven
|
||||
requeues; Step 11's manager-driven cases cover that. The Secret watch is
|
||||
wired in `SetupWithManager` but the label-restricted Secret cache it
|
||||
assumes arrives with `cmd/main.go` in Step 10.
|
||||
|
||||
## Step 5 — Health engine (`internal/health/`)
|
||||
|
||||
Implemented the engine per the plan's design: `probe.go` (through-the-proxy
|
||||
probe with the plan's exact transport — fresh per probe,
|
||||
`DisableKeepAlives: true` so every probe re-exercises CONNECT) and
|
||||
`engine.go` (leader-elected manager Runnable: 1 s scheduler tick + a pool
|
||||
of 8 workers, per-proxy threshold state under one mutex, transition-only
|
||||
emission over a buffered `chan event.GenericEvent`). The reconciler side
|
||||
landed in the same step: a `HealthSnapshotter` interface + `applyHealth`
|
||||
staging the Healthy condition/latency/lastHealthCheckTime from
|
||||
`Engine.Snapshot`, and a conditional
|
||||
`WatchesRawSource(source.Channel(...))` in `SetupWithManager`. Everything
|
||||
tolerates nil (engine unwired) until `cmd/main.go` connects the two in
|
||||
Step 10.
|
||||
|
||||
Deviations and judgment calls beyond the plan text:
|
||||
|
||||
- **First-probe scheduling is split by whether a verdict was seeded.** The
|
||||
plan's startup jitter (`nextDue = now + rand(0, interval)`) applies only
|
||||
to proxies whose state was seeded from an existing Healthy condition —
|
||||
the restart case it exists for. A never-probed proxy is probed on the
|
||||
next tick instead; making a brand-new proxy wait up to a full interval
|
||||
for its first verdict would be pure lag with no thundering-herd benefit.
|
||||
- **The latency-change emission rule only applies while the verdict is
|
||||
healthy.** Caught by the first test run, not foreseen: a success streak
|
||||
still below `successThreshold` (verdict unhealthy, reported unhealthy)
|
||||
satisfied the plan's rule (c) — latency delta vs a stale reported value,
|
||||
rate window open — and emitted a pointless latency-only update for a
|
||||
proxy still reported as unhealthy. Guarded with `res.ok && *st.healthy`.
|
||||
- **`ProbeTLSConfig` field added to the engine** (nil = system roots). The
|
||||
probe function needs a CA override to be testable against
|
||||
`httptest.NewTLSServer`, and the same knob is genuinely useful for
|
||||
probing targets signed by a private CA. Not a test-only backdoor.
|
||||
- **State pruning doubles as replacement hygiene:** any proxy with no
|
||||
probeable host (provisioning, mid-replacement, deleting) has its state
|
||||
dropped each tick, so a replacement instance always starts with fresh
|
||||
counters. Complementarily, the reconciler's create branch removes the
|
||||
stale Healthy condition and latency fields — a new VM shouldn't wear its
|
||||
predecessor's verdict.
|
||||
|
||||
Tests: a real CONNECT-capable proxy stub (hijack + bidirectional
|
||||
`io.Copy`) probing a real `httptest.NewTLSServer` — CONNECT success,
|
||||
refused CONNECT, unexpected status, dead proxy, plain-http forwarding;
|
||||
table-driven threshold/suppression/seeding/pruning tests driving
|
||||
`record`/`tick` directly; an end-to-end `Start` test (fake reader, fake
|
||||
probeFn, 5 ms tick) asserting event delivery, snapshot content, and clean
|
||||
shutdown on context cancel; and controller-side tests with a
|
||||
`fakeSnapshotter` proving Running+healthy ⇒ `Ready`, Running+unhealthy ⇒
|
||||
`Unhealthy`, no-verdict ⇒ no condition, and stale-verdict cleanup on
|
||||
replacement.
|
||||
|
||||
Verification (all green):
|
||||
|
||||
```bash
|
||||
make test # envtest + units; health 93.4%, controller 77.4%
|
||||
KUBEBUILDER_ASSETS="$PWD/bin/k8s/1.36.2-darwin-arm64" go test -race ./...
|
||||
go test -race -count=2 ./internal/health/ # shook out the emission-rule bug above
|
||||
```
|
||||
|
||||
Worth noting: `docs/architecture.md` (created between Steps 4 and 5 on
|
||||
user request) gained a §6 for the engine and now shows the third workqueue
|
||||
feed (`source.Channel`). The `Healthy` condition reasons live in the
|
||||
controller package (`ReasonProbeSucceeded`/`ReasonProbeFailed`) — the
|
||||
engine deliberately knows nothing about conditions except reading one at
|
||||
seed time, keeping the state/representation split honest. The
|
||||
`hint`-driven `wg.Go` idiom (Go 1.25+) replaced the classic
|
||||
`wg.Add/defer wg.Done` in the worker pool.
|
||||
|
||||
@@ -22,7 +22,7 @@ spec change deletes and recreates the VM. No in-place update logic.
|
||||
| API group | `crawl.example.com`, `v1alpha1`, kind `Proxy`, **namespaced** |
|
||||
| Git flow | Scaffold commit, then work on `feat/proxy-operator`, MR via `tea`, no merge/delete from CLI |
|
||||
| kubebuilder | `go install sigs.k8s.io/kubebuilder/v4/cmd/kubebuilder@v4.15.0` |
|
||||
| Mock health | Mock provider runs a **real in-process HTTP CONNECT proxy** per instance, so healthchecks genuinely pass and the kind demo is truly end-to-end |
|
||||
| Local/CI provider | **Kubernetes pod provider**, not an in-memory mock: creates real `ubuntu/squid` pods in-cluster. Revised mid-build — see Step 3 |
|
||||
| Verification | vet + unit + envtest, then a throwaway kind cluster running the README quickstart, then delete it |
|
||||
|
||||
### Verified environment — no version substitutions needed
|
||||
@@ -38,7 +38,7 @@ omitting the substitutions section.
|
||||
|
||||
### Milestone order — each ends green on `go build ./... && go vet ./...`
|
||||
|
||||
1. Scaffold + pins → 2. `api/v1alpha1` + CEL → 3. `internal/provider` + mock →
|
||||
1. Scaffold + pins → 2. `api/v1alpha1` + CEL → 3. `internal/provider` + kubernetes-pod →
|
||||
4. reconciler + envtest → 5. health engine → 6. lease + discovery → 7. GCP provider →
|
||||
8. orphan GC + metrics → 9. `cmd/main.go` wiring + `config/` → 10. docs + kind run.
|
||||
|
||||
@@ -64,8 +64,9 @@ plan to `docs/plans/2026-08-07-1747-proxy-operator.md` per CLAUDE.md. **Commit t
|
||||
untouched scaffold on its own** so every later diff is reviewable.
|
||||
|
||||
Post-scaffold hand-edits: `CONTROLLER_TOOLS_VERSION ?= v0.21.0` in the Makefile (CEL
|
||||
emission at the 1.36 API level); add a `run-mock` target; delete the scaffolded
|
||||
`.github/workflows/` (the remote is Gitea).
|
||||
emission at the 1.36 API level); add a `run-dev` target wired to a sample
|
||||
`--providers-config` (Step 10); delete the scaffolded `.github/workflows/` (the
|
||||
remote is Gitea).
|
||||
|
||||
---
|
||||
|
||||
@@ -120,12 +121,15 @@ health, discovery, and the hash.
|
||||
```
|
||||
internal/provider/{provider,errors,name,config,metrics}.go
|
||||
internal/provider/registry/registry.go # type→constructor — SEPARATE package
|
||||
internal/provider/{mock,gcp}/
|
||||
internal/provider/{kubernetes,gcp}/
|
||||
```
|
||||
|
||||
**Import-cycle trap:** a registry inside `internal/provider` would have to import
|
||||
`internal/provider/mock`, which imports `internal/provider`. `init()` self-registration
|
||||
is banned by CLAUDE.md, so the registry goes in its own leaf-importing package.
|
||||
`internal/provider/kubernetes` (or `.../gcp`), both of which import `internal/provider`
|
||||
for the interface. `init()` self-registration is banned by CLAUDE.md, so the registry
|
||||
goes in its own leaf-importing package, and its `Build` function takes the
|
||||
type→constructor map as a parameter instead — see Step 2's execution log entry for why
|
||||
this ended up better than a package-level map even beyond avoiding the cycle.
|
||||
|
||||
`Instance` needs **two fields the spec omits**, or orphan GC is unimplementable:
|
||||
`UID string` (from the label, for the liveness match) and `CreatedAt time.Time` (for
|
||||
@@ -157,23 +161,67 @@ same 80 bits. 80 bits → birthday collision at ~2^40 objects against a fleet of
|
||||
|
||||
---
|
||||
|
||||
## Step 3 — Mock provider (`internal/provider/mock/`)
|
||||
## Step 3 — Kubernetes pod provider (`internal/provider/kubernetes/`)
|
||||
|
||||
**State is a pure function of an injectable clock — no background timers.** `Get`/
|
||||
`ListByTag` derive state from `createdAt`/`deletedAt` vs `now()`:
|
||||
`< provisionDelay` → Provisioning; else Running; `deletedAt` set and `< deleteDelay` →
|
||||
Terminated; beyond that → purged, `ErrNotFound`. Deterministic under a fake clock,
|
||||
correct under the real one, and no goroutine lifecycle to leak.
|
||||
**Revised after Step 3 was first built as an in-memory mock provider** (state-machine
|
||||
simulation + a hand-rolled CONNECT proxy on a shared, refcounted local listener). The
|
||||
user found that too far from the real system to build confidence in, and didn't need
|
||||
tests to be fast enough to justify the complexity it cost — a real `kind`-cluster
|
||||
verification pass "once in a while" is an acceptable trade for tests that actually
|
||||
look like the final product. Full narrative of the reversal is in
|
||||
`docs/plans-executions/2026-08-07-1747-proxy-operator.md`; this section describes the
|
||||
replacement, which is what actually gets built for local dev/CI going forward. No
|
||||
in-memory provider remains in the tree — GCP is now the only other provider, per the
|
||||
spec's original two-provider scope.
|
||||
|
||||
**Real proxy listener (the user's decision):** when a record first reports `Running`,
|
||||
lazily start an `http.Server` on `127.0.0.1:0` implementing HTTP `CONNECT` tunnelling
|
||||
(hijack + bidirectional `io.Copy`) plus plain-HTTP forwarding, and report `127.0.0.1`
|
||||
plus the real listener port. `Delete` shuts it down. This is what makes the health
|
||||
engine's probe a genuine CONNECT through a real proxy, so the kind quickstart actually
|
||||
reaches Ready and leasable. One `http.Server` per instance, bounded by fleet size.
|
||||
**What it does:** `Create` creates a `corev1.Pod` running a proxy container in the
|
||||
same cluster (and same namespace as the owning Proxy CR — `req.Namespace`); `Get`
|
||||
reads the Pod's phase/IP; `Delete` deletes it (tolerating NotFound); `ListByTag` lists
|
||||
Pods by the standard `LabelManaged`/`LabelUID` labels, unscoped by namespace (the
|
||||
operator's RBAC needs cluster-scoped Pod permissions — see RBAC note below).
|
||||
|
||||
Fault injection: config-driven `failNextCreates`/`failWith` for the demo, plus
|
||||
`InjectCreateFailures(n int, class error)` for tests. `Create` is idempotent by name.
|
||||
**Proxy software: `ubuntu/squid`** (Canonical's actively maintained LTS image on
|
||||
Docker Hub, verified before picking it — 50M+ pulls, updated the same day this
|
||||
decision was made), not a hand-rolled proxy. It's a public image, so `kind` nodes
|
||||
pull it directly; no build/load step needed for the quickstart. Squid's config
|
||||
(`http_port <req.Port>`, permissive ACL) is generated in Go and injected via an env
|
||||
var the container's command writes to `/etc/squid/squid.conf` before exec'ing squid —
|
||||
no separate ConfigMap object, so there's still only one Kubernetes object per proxy
|
||||
instance to create, track, and clean up.
|
||||
|
||||
**providerID format: `<namespace>/<podName>`** (parseable with
|
||||
`k8s.io/client-go/tools/cache.SplitMetaNamespaceKey`), so `Get`/`Delete` are
|
||||
self-contained without needing to re-derive the namespace — the same reasoning as the
|
||||
GCP provider's zone-qualified providerID in Step 8.
|
||||
|
||||
**Pod naming:** reuses `provider.NameFromUID` unchanged — the same deterministic name
|
||||
satisfies Kubernetes Pod naming rules (`^[a-z0-9]([-a-z0-9]*[a-z0-9])?$`, ≤253 chars)
|
||||
with room to spare.
|
||||
|
||||
**State mapping:** Pod phase `Pending`, or `Running` with no PodIP yet → Provisioning
|
||||
(never publish an empty IP); `Running` with a PodIP → Running; `Succeeded`/`Failed`/
|
||||
`Unknown` → Terminated (the reconciler treats Stopped and Terminated identically —
|
||||
delete and recreate, cattle not pets — so collapsing three failure-ish phases into one
|
||||
is enough).
|
||||
|
||||
**Client:** built internally via `ctrl.GetConfig()` (auto-detects in-cluster config,
|
||||
falls back to the local kubeconfig otherwise), not threaded through the registry
|
||||
`Constructor` signature — this is what makes `make run` against a local `kind`
|
||||
cluster and running in-cluster use the exact same code path with no provider-specific
|
||||
wiring in `cmd/main.go`.
|
||||
|
||||
**Testing, given envtest can't schedule real Pods** (no kubelet — a Pod created
|
||||
against `envtest`'s API server just sits `Pending` forever): `internal/provider/kubernetes`
|
||||
itself is unit-tested against `sigs.k8s.io/controller-runtime/pkg/client/fake` — real
|
||||
Pod objects, real client interface, fully exercises Create/Get/Delete/ListByTag
|
||||
logic and the Pod-construction function in isolation, just without a kubelet actually
|
||||
starting a container. Real end-to-end proof (does a probe actually tunnel through a
|
||||
real Squid pod) only happens against a real `kind` cluster, in the Verification
|
||||
section — which is exactly what the user asked for. Step 4's reconciler tests use a
|
||||
small `Provider`-interface stub defined directly in the controller test file (a
|
||||
handful of lines, not a package) for exercising the state-machine's branching logic —
|
||||
categorically simpler than what mock.Provider was, since it has no config format, no
|
||||
fault-injection surface, and exists only inside test code.
|
||||
|
||||
---
|
||||
|
||||
@@ -425,14 +473,17 @@ registry → manager → `mgr.Add` health engine, GC, lease expiry loop, discove
|
||||
`SetupWithManager`. Contexts from `ctrl.SetupSignalHandler()` throughout.
|
||||
|
||||
RBAC markers: proxies CRUD + status + finalizers, secrets get/list/watch, events
|
||||
create/patch. `config/`: providers ConfigMap mount, `DISCOVERY_TOKEN` from a Secret,
|
||||
containerPort 8090 + Service. Samples: `proxy_mock.yaml`, `proxy_gcp.yaml`,
|
||||
`proxy_external.yaml`, `providers-config.yaml`.
|
||||
create/patch, and (for the kubernetes-pod provider) pods get/list/watch/create/delete
|
||||
— cluster-scoped, since `ListByTag` enumerates across namespaces. `config/`: providers
|
||||
ConfigMap mount, `DISCOVERY_TOKEN` from a Secret, containerPort 8090 + Service.
|
||||
Samples: `proxy_kubernetes.yaml`, `proxy_gcp.yaml`, `proxy_external.yaml`,
|
||||
`providers-config.yaml`.
|
||||
|
||||
`docs/architecture.md`: components table, ASCII data-flow diagram, and a **Decisions**
|
||||
section covering the channel-vs-patch choice, replacement-polls-to-NotFound, base32
|
||||
naming, discovery without leader election, single-mutex lease store, mock-runs-a-real-
|
||||
proxy, leader-handover health seeding, the latency-suppression refinement, the
|
||||
naming, discovery without leader election, single-mutex lease store, the mock→
|
||||
kubernetes-pod-provider revision (why, and why `ubuntu/squid` over a hand-rolled
|
||||
proxy), leader-handover health seeding, the latency-suppression refinement, the
|
||||
`lastHealthCheckTime` semantics, and the logr-not-slog deviation (CLAUDE.md says
|
||||
`slog`, but `log.FromContext(ctx)` returns logr inside controller paths — noted, not
|
||||
silently ignored).
|
||||
@@ -448,16 +499,17 @@ note confirming no substitutions were needed. Then a `CHANGELOG.md` entry with a
|
||||
|
||||
## Step 11 — Tests
|
||||
|
||||
**envtest** (`internal/controller/`), mock provider + a fake `HealthSnapshotter`,
|
||||
intervals shrunk to 50–200 ms, whole suite behind `testing.Short()`:
|
||||
Managed→Ready; spec change → old mock instance gone and providerID changed; delete →
|
||||
finalizer runs and instance removed; External → Ready on first health pass, no
|
||||
finalizer; injected quota → `Provisioned=False/QuotaExceeded` and phase *not* Failed;
|
||||
permanent error → Failed and no further provider calls; adopt (strip annotation →
|
||||
restored, providerID unchanged); and the **CEL cases only a real API server can test** —
|
||||
mode/provider mutation rejected, Managed-without-provider, External-without-endpoint,
|
||||
cloudInit both/neither, and `healthCheck` omitted → nested defaults materialized (the
|
||||
`default={}` assertion).
|
||||
**envtest** (`internal/controller/`), a small in-test stub `Provider` (not the real
|
||||
kubernetes-pod provider — envtest has no kubelet, so a real Pod never leaves Pending)
|
||||
plus a fake `HealthSnapshotter`, intervals shrunk to 50–200 ms, whole suite behind
|
||||
`testing.Short()`: Managed→Ready; spec change → old stub instance gone and
|
||||
providerID changed; delete → finalizer runs and instance removed; External → Ready
|
||||
on first health pass, no finalizer; injected quota → `Provisioned=False/QuotaExceeded`
|
||||
and phase *not* Failed; permanent error → Failed and no further provider calls; adopt
|
||||
(strip annotation → restored, providerID unchanged); and the **CEL cases only a real
|
||||
API server can test** — mode/provider mutation rejected, Managed-without-provider,
|
||||
External-without-endpoint, cloudInit both/neither, and `healthCheck` omitted → nested
|
||||
defaults materialized (the `default={}` assertion).
|
||||
|
||||
**Action-table unit tests** — the highest-value tests in the repo: `fake` client with
|
||||
`WithStatusSubresource`, calling `Reconcile` directly, table-driven over every row of
|
||||
@@ -466,12 +518,15 @@ client runs neither CEL nor defaulting — that's what the envtest CEL cases cov
|
||||
|
||||
**Units:** name derivation (idempotency, `^proxy-[a-z2-7]{16}$`, 10k-UID distinctness);
|
||||
`Class()` mapping + `errors.Is`/`errors.As` through the multi-unwrap; config loading;
|
||||
mock state at the delay boundary with a fake clock; `buildInsertRequest` field-by-field
|
||||
+ GCP error classification + RUNNING-without-IP; `computePhase` truth table; `SpecHash`
|
||||
stability *and* sensitivity; lease store (capacity, `MaxLeases=0`, least-loaded with
|
||||
latency tie-break, cooldown with/without target, report on an expired-but-retained
|
||||
lease, concurrent acquire under `-race` never exceeding `MaxLeases`); discovery
|
||||
handlers over `httptest` + fake reader + real store; health thresholds against a real
|
||||
kubernetes-pod provider Create/Get/Delete/ListByTag against
|
||||
`sigs.k8s.io/controller-runtime/pkg/client/fake` (real Pod objects, real client
|
||||
interface, no kubelet needed for this level) plus pure tests of the generated Squid
|
||||
config and Pod spec; `buildInsertRequest` field-by-field + GCP error classification +
|
||||
RUNNING-without-IP; `computePhase` truth table; `SpecHash` stability *and*
|
||||
sensitivity; lease store (capacity, `MaxLeases=0`, least-loaded with latency
|
||||
tie-break, cooldown with/without target, report on an expired-but-retained lease,
|
||||
concurrent acquire under `-race` never exceeding `MaxLeases`); discovery handlers
|
||||
over `httptest` + fake reader + real store; health thresholds against a real
|
||||
CONNECT-capable `httptest` proxy stub.
|
||||
|
||||
Everything runs with `-race`.
|
||||
@@ -485,19 +540,20 @@ go vet ./... && make test && make build # unit + envtest, -race
|
||||
|
||||
kind create cluster --name proxy-operator-demo
|
||||
make install
|
||||
make run-mock & # --providers-config hack/providers-mock.yaml
|
||||
kubectl apply -f config/samples/proxy_mock.yaml
|
||||
kubectl get px -w # expect Ready with an IP
|
||||
make run-dev & # --providers-config hack/providers-dev.yaml
|
||||
kubectl apply -f config/samples/proxy_kubernetes.yaml
|
||||
kubectl get px -w # expect Ready with an IP (a real squid Pod)
|
||||
curl -s 'localhost:8090/v1/proxies?healthy=true' | jq
|
||||
curl -s -XPOST localhost:8090/v1/leases -d '{"selector":{"geo":"eu"},"ttlSeconds":300}' | jq
|
||||
curl -s -XPOST localhost:8090/v1/leases/<id>/report -d '{"result":"rate_limited","target":"example.com"}'
|
||||
curl -si -XDELETE localhost:8090/v1/leases/<id> # 204, and 204 again
|
||||
kubectl delete -f config/samples/proxy_mock.yaml # finalizer runs, object goes
|
||||
kubectl delete -f config/samples/proxy_kubernetes.yaml # finalizer runs, Pod is deleted
|
||||
kind delete cluster --name proxy-operator-demo
|
||||
```
|
||||
|
||||
Success bar: a competent SRE clones the repo, follows the README, and holds a lease on a
|
||||
healthy mock proxy in under 10 minutes.
|
||||
healthy proxy — a real Squid pod running in their own `kind` cluster — in under 10
|
||||
minutes.
|
||||
|
||||
Then commit on `feat/proxy-operator`, push with `-u`, open the MR with
|
||||
`tea pr create --base main --head feat/proxy-operator`, print the URL. No merging or
|
||||
|
||||
128
docs/plans/2026-08-08-1335-lean-scaffold-cleanup.md
Normal file
128
docs/plans/2026-08-08-1335-lean-scaffold-cleanup.md
Normal file
@@ -0,0 +1,128 @@
|
||||
# Plan: Lean-down cleanup — strip non-goal scaffold from the application footprint
|
||||
**Created:** 2026-08-08 13:35
|
||||
|
||||
## Context
|
||||
|
||||
The user asked two things: (1) explain why webhook-related pieces existed at all when
|
||||
the spec ([docs/prompts/__initial-prompt.md](docs/prompts/__initial-prompt.md) §12)
|
||||
says "do NOT build: admission webhooks, cert-manager wiring", and (2) audit all
|
||||
completed work (Steps 0–3) for anything extra, so the project starts as lean as
|
||||
possible before Step 4 (reconciler) begins.
|
||||
|
||||
**The answer to (1), already given in conversation and to be recorded in the
|
||||
execution log:** `kubebuilder init` unconditionally generates webhook machinery for
|
||||
every project. At Step 0 the scaffold was deliberately committed untouched as a
|
||||
reviewable baseline, with only `.github/` stripped; spec §3's "config/
|
||||
(scaffold-generated, kept working)" was read as license to keep the rest. That was a
|
||||
process gap — §12's non-goals deserved an active pruning pass immediately after the
|
||||
baseline commit, especially for parts that actually *did* something (main.go started
|
||||
a real webhook server; e2e installed cert-manager). The active parts were already
|
||||
removed in commit `7700358` after the user noticed; this plan removes what remains.
|
||||
|
||||
**Could `init` have been told to skip it? No — verified against the v4.15.0 binary
|
||||
(`kubebuilder init --help`), not from memory.** Its full flag surface is
|
||||
domain/repo/owner/license/multigroup/namespaced/fetch-deps/skip-go-version-check/
|
||||
project-version/plugins; nothing subtracts features. Plugins are purely additive
|
||||
(helm, grafana, deploy-image, autoupdate — no "minimal" plugin), and while webhook
|
||||
*code* only appears via `create webhook` (never run here), the baseline *plumbing*
|
||||
(main.go webhook server, commented kustomize blocks, cert-manager in e2e utils,
|
||||
prometheus/network-policy dirs) is emitted unconditionally so later `create webhook`
|
||||
runs have anchors. Scaffold-then-prune is the only supported path to a lean
|
||||
baseline. (`--namespaced` was the one arguably-applicable flag, but it conflicts
|
||||
with the kubernetes-pod provider's cluster-wide `ListByTag` for orphan GC —
|
||||
cluster-scoped was correct.) Record this in the execution log so the next project
|
||||
bootstrap knows to plan a pruning pass at scaffold time.
|
||||
|
||||
**Scope, per the user's explicit direction:** "tooling around the project is cool, i
|
||||
just want the application code produced to start as lean as possible." So developer
|
||||
tooling stays untouched — `.golangci.yml`, `.custom-gcl.yml`, `.devcontainer/`,
|
||||
`AGENTS.md`, Makefile `lint`/`docker-buildx`/`build-installer` targets, and
|
||||
`.claude/settings.json` are all explicitly KEPT. The cleanup targets only the
|
||||
application and its deployed footprint: `config/` manifests that `make deploy`
|
||||
would apply or that exist solely to serve never-to-be-built features.
|
||||
|
||||
**Code-level audit result (part of this task's deliverable, no action needed):**
|
||||
Steps 1–3's Go code contains nothing beyond spec that isn't a justified,
|
||||
already-logged deviation (`Instance.UID`/`CreatedAt` for orphan GC, `MaxLeases
|
||||
*int32`, `HealthCheck default={}`, registry-as-parameter, kubernetes provider
|
||||
replacing mock per user decision). Trivial extras (`shortName=px`,
|
||||
`MaxProperties=32` on attributes) are harmless and stay. `metrics_auth_role*.yaml` /
|
||||
`metrics_reader_role.yaml` are genuinely used by the secure-metrics filter and the
|
||||
e2e metrics test — they stay.
|
||||
|
||||
## Scope refinements from user review
|
||||
|
||||
- **KEEP `config/prometheus/`** entirely (user: monitoring manifests stay),
|
||||
including `monitor_tls_patch.yaml` and the commented `#- ../prometheus` enable
|
||||
line in the default kustomization.
|
||||
- **KEEP the paired metrics-TLS plumbing** for coherence with the kept prometheus
|
||||
TLS patch: `config/default/cert_metrics_manager_patch.yaml`, its commented
|
||||
`[METRICS-WITH-CERTS]` reference, and the *metrics-certs/ServiceMonitor halves*
|
||||
of the commented replacements block. Removing half of a pair would leave
|
||||
dangling comment references — mess, not lean.
|
||||
- **REMOVE `config/network-policy/`** (user: "we do not need any network policies
|
||||
at the moment").
|
||||
- **KEEP all RBAC manifests** (user: "i want to keep manifests relevant to
|
||||
rbacs") — including the `proxy_admin/editor/viewer` helper ClusterRoles
|
||||
originally slated for removal.
|
||||
|
||||
## Removals (all in `config/`)
|
||||
|
||||
1. **`config/network-policy/`** (2 files: kustomization.yaml,
|
||||
allow-metrics-traffic.yaml) + the commented `#- ../network-policy` line and its
|
||||
`[NETWORK POLICY]` banner in `config/default/kustomization.yaml`.
|
||||
|
||||
2. **`config/default/kustomization.yaml`** — strip the *webhook-only* parts: the
|
||||
commented `#- ../webhook` and `#- ../certmanager` resource lines with their
|
||||
banners; the commented `manager_webhook_patch.yaml` patch reference; and the
|
||||
webhook halves of the commented replacements block (`serving-cert` Certificate
|
||||
sources targeting Validating/Mutating WebhookConfiguration cainjection, the
|
||||
conversion-webhook block, and the
|
||||
`+kubebuilder:scaffold:crdkustomizecainjectionns`/`...name` markers — anchors
|
||||
only for `kubebuilder create webhook`, which will never run here). The
|
||||
metrics-certs/ServiceMonitor replacement halves stay (see scope refinements).
|
||||
|
||||
3. **`config/crd/kustomization.yaml`** — strip the two commented `[WEBHOOK]` blocks
|
||||
(conversion-webhook patches and the `configurations:` reference) and the
|
||||
`+kubebuilder:scaffold:crdkustomizewebhookpatch` marker. **Keep** the
|
||||
`+kubebuilder:scaffold:crdkustomizeresource` marker (one line; anchors
|
||||
`kubebuilder create api`, which could legitimately run again).
|
||||
|
||||
4. **`config/crd/kustomizeconfig.yaml`** — teaches kustomize how to rewrite webhook
|
||||
conversion service references; only consumer was the commented block in (3).
|
||||
|
||||
`config/rbac/` is untouched: the operator's own role/bindings, the metrics-auth
|
||||
roles, *and* the `proxy_admin/editor/viewer` helper ClusterRoles all stay per the
|
||||
user's direction.
|
||||
|
||||
## Execution steps
|
||||
|
||||
1. Save this plan into the repo per CLAUDE.md: `docs/plans/$(date
|
||||
"+%Y-%m-%d-%H%M")-lean-scaffold-cleanup.md`, committed on its own before the
|
||||
cleanup work starts.
|
||||
2. Delete the files listed above (`git rm`); edit the two kustomization.yaml files.
|
||||
3. Verify:
|
||||
- `bin/kustomize build config/default` renders cleanly (proves no dangling
|
||||
references; kustomize is already in `bin/` from the scaffold).
|
||||
- `bin/kustomize build config/crd` renders cleanly.
|
||||
- `go build ./... && go vet ./...` and the `-tags=e2e` variants stay clean.
|
||||
- `make test` stays green.
|
||||
4. Append an execution-log section to
|
||||
[docs/plans-executions/2026-08-07-1747-proxy-operator.md](docs/plans-executions/2026-08-07-1747-proxy-operator.md)
|
||||
covering: the "why webhooks existed" explanation (scaffold origin + the Step 0
|
||||
process gap), the audit's clean bill for the Go code, the keep-tooling scope
|
||||
decision, and the exact removals.
|
||||
5. Single commit on `feat/proxy-operator` with the plan-file commit preceding it.
|
||||
|
||||
## Verification
|
||||
|
||||
```bash
|
||||
bin/kustomize build config/default > /dev/null && echo default-ok
|
||||
bin/kustomize build config/crd > /dev/null && echo crd-ok
|
||||
go build ./... && go vet ./...
|
||||
go build -tags=e2e ./... && go vet -tags=e2e ./...
|
||||
make test
|
||||
```
|
||||
|
||||
All must pass with output identical in substance to pre-cleanup (same tests, same
|
||||
coverage numbers).
|
||||
2
go.mod
2
go.mod
@@ -5,6 +5,7 @@ go 1.26.0
|
||||
require (
|
||||
github.com/onsi/ginkgo/v2 v2.27.4
|
||||
github.com/onsi/gomega v1.39.0
|
||||
k8s.io/api v0.36.0
|
||||
k8s.io/apimachinery v0.36.0
|
||||
k8s.io/client-go v0.36.0
|
||||
sigs.k8s.io/controller-runtime v0.24.1
|
||||
@@ -85,7 +86,6 @@ require (
|
||||
gopkg.in/evanphx/json-patch.v4 v4.13.0 // indirect
|
||||
gopkg.in/inf.v0 v0.9.1 // indirect
|
||||
gopkg.in/yaml.v3 v3.0.1 // indirect
|
||||
k8s.io/api v0.36.0 // indirect
|
||||
k8s.io/apiextensions-apiserver v0.36.0 // indirect
|
||||
k8s.io/apiserver v0.36.0 // indirect
|
||||
k8s.io/component-base v0.36.0 // indirect
|
||||
|
||||
175
internal/controller/health_test.go
Normal file
175
internal/controller/health_test.go
Normal file
@@ -0,0 +1,175 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/health"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
type fakeSnapshotter struct {
|
||||
snap health.Snapshot
|
||||
ok bool
|
||||
}
|
||||
|
||||
func (f fakeSnapshotter) Snapshot(types.NamespacedName) (health.Snapshot, bool) {
|
||||
return f.snap, f.ok
|
||||
}
|
||||
|
||||
// TestReconcile_healthRepresentation covers the reconciler's half of the
|
||||
// health split: turning the engine's Snapshot into the Healthy condition,
|
||||
// the latency fields, and ultimately the Ready/Unhealthy phases.
|
||||
func TestReconcile_healthRepresentation(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
probeTime := time.Now()
|
||||
freshHash := specHash(managedProxy(), "")
|
||||
runningStub := func() *stubProvider {
|
||||
return &stubProvider{
|
||||
getInst: &provider.Instance{ID: "stub-id-1", IP: "10.1.2.3", State: provider.StateRunning},
|
||||
}
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
proxy *crawlv1alpha1.Proxy
|
||||
stub *stubProvider
|
||||
health HealthSnapshotter
|
||||
wantPhase crawlv1alpha1.ProxyPhase
|
||||
verify func(t *testing.T, r *ProxyReconciler)
|
||||
}{
|
||||
{
|
||||
name: "running and healthy becomes Ready",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: runningStub(),
|
||||
health: fakeSnapshotter{ok: true, snap: health.Snapshot{
|
||||
Healthy: true, Latency: 37 * time.Millisecond, LastProbe: probeTime,
|
||||
}},
|
||||
wantPhase: crawlv1alpha1.PhaseReady,
|
||||
verify: func(t *testing.T, r *ProxyReconciler) {
|
||||
p := getProxy(t, r)
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionHealthy, metav1.ConditionTrue, ReasonProbeSucceeded)
|
||||
if p.Status.LatencyMillis != 37 {
|
||||
t.Errorf("latencyMillis = %d, want 37", p.Status.LatencyMillis)
|
||||
}
|
||||
if p.Status.LastHealthCheckTime == nil {
|
||||
t.Error("lastHealthCheckTime not set")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "running but unhealthy becomes Unhealthy",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: runningStub(),
|
||||
health: fakeSnapshotter{ok: true, snap: health.Snapshot{
|
||||
Healthy: false, LastProbe: probeTime,
|
||||
LastError: "CONNECT refused", ConsecutiveFailures: 3,
|
||||
}},
|
||||
wantPhase: crawlv1alpha1.PhaseUnhealthy,
|
||||
verify: func(t *testing.T, r *ProxyReconciler) {
|
||||
p := getProxy(t, r)
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionHealthy, metav1.ConditionFalse, ReasonProbeFailed)
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy)
|
||||
if !strings.Contains(cond.Message, "CONNECT refused") {
|
||||
t.Errorf("condition message %q does not carry the probe error", cond.Message)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "no verdict yet stays Provisioning without a Healthy condition",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: runningStub(),
|
||||
health: fakeSnapshotter{ok: false},
|
||||
wantPhase: crawlv1alpha1.PhaseProvisioning,
|
||||
verify: func(t *testing.T, r *ProxyReconciler) {
|
||||
p := getProxy(t, r)
|
||||
if apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy) != nil {
|
||||
t.Error("Healthy condition present without an engine verdict")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "external proxy with a healthy verdict becomes Ready",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Finalizers = nil
|
||||
p.Spec = crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "203.0.113.7"},
|
||||
}
|
||||
}),
|
||||
stub: &stubProvider{},
|
||||
health: fakeSnapshotter{ok: true, snap: health.Snapshot{
|
||||
Healthy: true, Latency: 5 * time.Millisecond, LastProbe: probeTime,
|
||||
}},
|
||||
wantPhase: crawlv1alpha1.PhaseReady,
|
||||
verify: func(t *testing.T, r *ProxyReconciler) {
|
||||
assertCondition(t, getProxy(t, r), crawlv1alpha1.ConditionHealthy, metav1.ConditionTrue, ReasonProbeSucceeded)
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "creating a replacement clears the stale Healthy verdict",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Status.Conditions = []metav1.Condition{{
|
||||
Type: crawlv1alpha1.ConditionHealthy, Status: metav1.ConditionTrue,
|
||||
Reason: ReasonProbeSucceeded, LastTransitionTime: metav1.Now(),
|
||||
}}
|
||||
p.Status.LatencyMillis = 42
|
||||
p.Status.LastHealthCheckTime = &metav1.Time{Time: probeTime}
|
||||
}),
|
||||
stub: &stubProvider{createID: "stub-id-2"},
|
||||
health: fakeSnapshotter{ok: false},
|
||||
wantPhase: crawlv1alpha1.PhaseProvisioning,
|
||||
verify: func(t *testing.T, r *ProxyReconciler) {
|
||||
p := getProxy(t, r)
|
||||
if apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy) != nil {
|
||||
t.Error("stale Healthy condition survived instance creation")
|
||||
}
|
||||
if p.Status.LatencyMillis != 0 || p.Status.LastHealthCheckTime != nil {
|
||||
t.Error("stale latency fields survived instance creation")
|
||||
}
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
r := newTestReconciler(t, tc.stub, tc.proxy)
|
||||
r.Health = tc.health
|
||||
if _, err := doReconcile(t, r); err != nil {
|
||||
t.Fatalf("Reconcile: %v", err)
|
||||
}
|
||||
if got := getProxy(t, r).Status.Phase; got != tc.wantPhase {
|
||||
t.Errorf("phase = %s, want %s", got, tc.wantPhase)
|
||||
}
|
||||
tc.verify(t, r)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A nil Health snapshotter must disable representation entirely.
|
||||
func TestReconcile_nilHealthSnapshotter(t *testing.T) {
|
||||
t.Parallel()
|
||||
freshHash := specHash(managedProxy(), "")
|
||||
r := newTestReconciler(t,
|
||||
&stubProvider{getInst: &provider.Instance{ID: "stub-id-1", IP: "10.1.2.3", State: provider.StateRunning}},
|
||||
managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)))
|
||||
|
||||
if _, err := doReconcile(t, r); err != nil {
|
||||
t.Fatalf("Reconcile: %v", err)
|
||||
}
|
||||
p := getProxy(t, r)
|
||||
if apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy) != nil {
|
||||
t.Error("Healthy condition written with no snapshotter configured")
|
||||
}
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseProvisioning {
|
||||
t.Errorf("phase = %s, want Provisioning", p.Status.Phase)
|
||||
}
|
||||
}
|
||||
@@ -18,46 +18,411 @@ package controller
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/controller"
|
||||
"sigs.k8s.io/controller-runtime/pkg/controller/controllerutil"
|
||||
"sigs.k8s.io/controller-runtime/pkg/event"
|
||||
"sigs.k8s.io/controller-runtime/pkg/handler"
|
||||
logf "sigs.k8s.io/controller-runtime/pkg/log"
|
||||
"sigs.k8s.io/controller-runtime/pkg/reconcile"
|
||||
"sigs.k8s.io/controller-runtime/pkg/source"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/health"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// ProxyReconciler reconciles a Proxy object
|
||||
// HealthSnapshotter provides the current probe verdict for a proxy. The
|
||||
// health engine implements it; the reconciler is its only consumer, turning
|
||||
// snapshots into the Healthy condition — the engine owns health state, the
|
||||
// reconciler owns its representation.
|
||||
type HealthSnapshotter interface {
|
||||
Snapshot(key types.NamespacedName) (health.Snapshot, bool)
|
||||
}
|
||||
|
||||
// ProxyReconciler reconciles Proxy objects as a state machine: every
|
||||
// reconcile derives exactly one action from (spec, status, provider Get),
|
||||
// performs it, and requeues. Status is written at most once per reconcile,
|
||||
// by the deferred patch in Reconcile.
|
||||
type ProxyReconciler struct {
|
||||
client.Client
|
||||
Scheme *runtime.Scheme
|
||||
|
||||
// Providers maps spec.provider values to configured backends.
|
||||
Providers map[string]provider.Provider
|
||||
|
||||
// Health supplies probe verdicts; nil disables health representation
|
||||
// (the Healthy condition simply never appears).
|
||||
Health HealthSnapshotter
|
||||
// HealthEvents, when non-nil, is watched as a raw source so the health
|
||||
// engine can enqueue proxies on status-affecting transitions.
|
||||
HealthEvents <-chan event.GenericEvent
|
||||
|
||||
// Poll intervals are struct fields, never consts, so tests can shrink
|
||||
// them to milliseconds.
|
||||
ProvisioningPoll time.Duration // while waiting for an instance to reach Running
|
||||
DriftPoll time.Duration // between re-checks of a Running instance
|
||||
DeletionPoll time.Duration // while waiting for an instance to disappear
|
||||
QuotaRetry time.Duration // after ErrQuotaExceeded; slow, off the backoff curve
|
||||
RequeueNow time.Duration // "process the next state promptly" (Result.Requeue is deprecated)
|
||||
}
|
||||
|
||||
// +kubebuilder:rbac:groups=crawl.example.com,resources=proxies,verbs=get;list;watch;create;update;patch;delete
|
||||
// +kubebuilder:rbac:groups=crawl.example.com,resources=proxies/status,verbs=get;update;patch
|
||||
// +kubebuilder:rbac:groups=crawl.example.com,resources=proxies/finalizers,verbs=update
|
||||
// +kubebuilder:rbac:groups="",resources=secrets,verbs=get;list;watch
|
||||
|
||||
// Reconcile is part of the main kubernetes reconciliation loop which aims to
|
||||
// move the current state of the cluster closer to the desired state.
|
||||
// TODO(user): Modify the Reconcile function to compare the state specified by
|
||||
// the Proxy object against the actual cluster state, and then
|
||||
// perform operations to make the cluster state reflect the state specified by
|
||||
// the user.
|
||||
//
|
||||
// For more details, check Reconcile and its Result here:
|
||||
// - https://pkg.go.dev/sigs.k8s.io/controller-runtime@v0.24.1/pkg/reconcile
|
||||
func (r *ProxyReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) {
|
||||
_ = logf.FromContext(ctx)
|
||||
// Reconcile fetches the Proxy named by req into p (r.Get fills the struct
|
||||
// through the pointer), dispatches to the delete/external/managed state
|
||||
// machines, and flushes any status change exactly once on the way out.
|
||||
func (r *ProxyReconciler) Reconcile(ctx context.Context, req ctrl.Request) (res ctrl.Result, err error) {
|
||||
var p crawlv1alpha1.Proxy
|
||||
if err := r.Get(ctx, req.NamespacedName, &p); err != nil {
|
||||
return ctrl.Result{}, client.IgnoreNotFound(err)
|
||||
}
|
||||
base := p.DeepCopy()
|
||||
defer func() {
|
||||
// NotFound is expected when this reconcile just removed the last
|
||||
// finalizer and the object is already gone.
|
||||
if perr := r.patchStatusIfChanged(ctx, base, &p); perr != nil && !apierrors.IsNotFound(perr) {
|
||||
err = errors.Join(err, perr)
|
||||
}
|
||||
}()
|
||||
|
||||
// TODO(user): your logic here
|
||||
switch {
|
||||
case !p.DeletionTimestamp.IsZero():
|
||||
return r.reconcileDelete(ctx, &p)
|
||||
case p.Spec.Mode == crawlv1alpha1.ModeExternal:
|
||||
return r.reconcileExternal(ctx, &p)
|
||||
default:
|
||||
return r.reconcileManaged(ctx, &p)
|
||||
}
|
||||
}
|
||||
|
||||
func (r *ProxyReconciler) reconcileManaged(ctx context.Context, p *crawlv1alpha1.Proxy) (ctrl.Result, error) {
|
||||
log := logf.FromContext(ctx)
|
||||
|
||||
if controllerutil.AddFinalizer(p, crawlv1alpha1.FinalizerName) {
|
||||
// The Update event re-triggers reconciliation; provisioning starts
|
||||
// on the next pass, with the finalizer safely persisted first.
|
||||
return ctrl.Result{}, r.Update(ctx, p)
|
||||
}
|
||||
|
||||
// Permanent-failure latch: once this generation has failed permanently,
|
||||
// stop calling the provider until the spec changes.
|
||||
if cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned); cond != nil &&
|
||||
cond.Status == metav1.ConditionFalse && cond.Reason == ReasonPermanentError &&
|
||||
cond.ObservedGeneration == p.Generation {
|
||||
return ctrl.Result{}, nil
|
||||
}
|
||||
|
||||
// SetupWithManager sets up the controller with the Manager.
|
||||
prov, ok := r.Providers[p.Spec.Provider]
|
||||
if !ok {
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonPermanentError,
|
||||
fmt.Sprintf("provider %q is not configured", p.Spec.Provider))
|
||||
return ctrl.Result{}, nil
|
||||
}
|
||||
|
||||
cloudInit, err := r.resolveCloudInit(ctx, p)
|
||||
if err != nil {
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonCloudInitError, err.Error())
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
hash := specHash(p, cloudInit)
|
||||
|
||||
if p.Status.ProviderID == "" {
|
||||
id, err := prov.Create(ctx, provider.CreateRequest{
|
||||
Name: provider.NameFromUID(p.UID),
|
||||
UID: string(p.UID),
|
||||
Namespace: p.Namespace,
|
||||
ProxyName: p.Name,
|
||||
Placement: placementFrom(p.Spec.Placement),
|
||||
CloudInit: cloudInit,
|
||||
Port: p.EffectivePort(),
|
||||
})
|
||||
if err != nil {
|
||||
return r.providerFailure(p, err)
|
||||
}
|
||||
log.Info("created instance", "provider", p.Spec.Provider, "providerID", id)
|
||||
if err := r.setSpecHash(ctx, p, hash); err != nil {
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
p.Status.ProviderID = id
|
||||
p.Status.IP = ""
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonProvisioning, "instance created; waiting for it to run")
|
||||
// Any Healthy verdict belonged to the previous instance; the health
|
||||
// engine starts fresh for the new one (its state was pruned while
|
||||
// the proxy had no IP), and so must the status.
|
||||
apimeta.RemoveStatusCondition(&p.Status.Conditions, crawlv1alpha1.ConditionHealthy)
|
||||
p.Status.LatencyMillis = 0
|
||||
p.Status.LastHealthCheckTime = nil
|
||||
return ctrl.Result{RequeueAfter: r.ProvisioningPoll}, nil
|
||||
}
|
||||
|
||||
if ann := p.Annotations[crawlv1alpha1.AnnotationSpecHash]; ann != hash {
|
||||
if ann == "" {
|
||||
// Adopt: an instance provisioned before the hash-input struct
|
||||
// gained a field (or by an older operator version) keeps its
|
||||
// instance; replacing the whole fleet on upgrade would be wrong.
|
||||
if err := r.setSpecHash(ctx, p, hash); err != nil {
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
return ctrl.Result{RequeueAfter: r.RequeueNow}, nil
|
||||
}
|
||||
return r.replaceInstance(ctx, p, prov, hash)
|
||||
}
|
||||
|
||||
inst, err := prov.Get(ctx, p.Status.ProviderID)
|
||||
if provider.Class(err) == provider.ErrNotFound {
|
||||
p.Status.ProviderID = ""
|
||||
p.Status.IP = ""
|
||||
return ctrl.Result{RequeueAfter: r.RequeueNow}, nil
|
||||
}
|
||||
if err != nil {
|
||||
return r.providerFailure(p, err)
|
||||
}
|
||||
|
||||
switch inst.State {
|
||||
case provider.StateProvisioning:
|
||||
p.Status.IP = ""
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonProvisioning, "waiting for the instance to run")
|
||||
return ctrl.Result{RequeueAfter: r.ProvisioningPoll}, nil
|
||||
case provider.StateRunning:
|
||||
p.Status.IP = inst.IP
|
||||
setProvisioned(p, metav1.ConditionTrue, ReasonCreated, "instance is running")
|
||||
r.applyHealth(p)
|
||||
return ctrl.Result{RequeueAfter: r.DriftPoll}, nil
|
||||
default: // Stopped, Terminated: cattle, not pets — delete and recreate.
|
||||
if err := prov.Delete(ctx, p.Status.ProviderID); err != nil {
|
||||
return r.providerFailure(p, err)
|
||||
}
|
||||
log.Info("deleting instance for recreation", "providerID", p.Status.ProviderID, "state", inst.State)
|
||||
p.Status.IP = ""
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonRecreating,
|
||||
fmt.Sprintf("instance is %s; deleting it for recreation", inst.State))
|
||||
return ctrl.Result{RequeueAfter: r.DeletionPoll}, nil
|
||||
}
|
||||
}
|
||||
|
||||
// replaceInstance handles a spec-hash mismatch. The replacement instance has
|
||||
// the same deterministic name as the old one (both derive from the CR UID),
|
||||
// so recreating before the old instance is fully gone would hit "already
|
||||
// exists" — hence: delete, poll to NotFound, only then advance the hash and
|
||||
// let the create branch run.
|
||||
func (r *ProxyReconciler) replaceInstance(ctx context.Context, p *crawlv1alpha1.Proxy, prov provider.Provider, hash string) (ctrl.Result, error) {
|
||||
_, err := prov.Get(ctx, p.Status.ProviderID)
|
||||
if provider.Class(err) == provider.ErrNotFound {
|
||||
// Old instance is gone. The Update inside setSpecHash refreshes p
|
||||
// from the server — including status — so the status clear must be
|
||||
// staged after it, or it would be silently overwritten. A crash
|
||||
// between the two writes recovers either way: the create branch's
|
||||
// Create is idempotent by name, and a stale ID resolves to NotFound
|
||||
// again.
|
||||
if err := r.setSpecHash(ctx, p, hash); err != nil {
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
p.Status.ProviderID = ""
|
||||
p.Status.IP = ""
|
||||
return ctrl.Result{RequeueAfter: r.RequeueNow}, nil
|
||||
}
|
||||
if err != nil {
|
||||
return r.providerFailure(p, err)
|
||||
}
|
||||
if err := prov.Delete(ctx, p.Status.ProviderID); err != nil {
|
||||
return r.providerFailure(p, err)
|
||||
}
|
||||
logf.FromContext(ctx).Info("replacing instance after spec change", "providerID", p.Status.ProviderID)
|
||||
p.Status.IP = ""
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonReplacing, "spec changed; deleting the old instance before recreating")
|
||||
return ctrl.Result{RequeueAfter: r.DeletionPoll}, nil
|
||||
}
|
||||
|
||||
func (r *ProxyReconciler) reconcileDelete(ctx context.Context, p *crawlv1alpha1.Proxy) (ctrl.Result, error) {
|
||||
if !controllerutil.ContainsFinalizer(p, crawlv1alpha1.FinalizerName) {
|
||||
return ctrl.Result{}, nil
|
||||
}
|
||||
if p.Status.ProviderID == "" {
|
||||
// Nothing was ever recorded as created; orphan GC reaps any stray
|
||||
// instance a crashed create might have left behind.
|
||||
controllerutil.RemoveFinalizer(p, crawlv1alpha1.FinalizerName)
|
||||
return ctrl.Result{}, r.Update(ctx, p)
|
||||
}
|
||||
prov, ok := r.Providers[p.Spec.Provider]
|
||||
if !ok {
|
||||
return ctrl.Result{}, fmt.Errorf(
|
||||
"provider %q is not configured; cannot clean up instance %s", p.Spec.Provider, p.Status.ProviderID)
|
||||
}
|
||||
_, err := prov.Get(ctx, p.Status.ProviderID)
|
||||
if provider.Class(err) == provider.ErrNotFound {
|
||||
controllerutil.RemoveFinalizer(p, crawlv1alpha1.FinalizerName)
|
||||
return ctrl.Result{}, r.Update(ctx, p)
|
||||
}
|
||||
if err != nil {
|
||||
return r.deletionFailure(err)
|
||||
}
|
||||
if err := prov.Delete(ctx, p.Status.ProviderID); err != nil {
|
||||
return r.deletionFailure(err)
|
||||
}
|
||||
logf.FromContext(ctx).Info("deleting instance", "providerID", p.Status.ProviderID)
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonDeleting, "deleting the instance before removing the finalizer")
|
||||
return ctrl.Result{RequeueAfter: r.DeletionPoll}, nil
|
||||
}
|
||||
|
||||
func (r *ProxyReconciler) reconcileExternal(_ context.Context, p *crawlv1alpha1.Proxy) (ctrl.Result, error) {
|
||||
if p.Spec.Endpoint == nil {
|
||||
// CEL guarantees an endpoint on any object that went through the API
|
||||
// server; tolerate its absence instead of panicking.
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonPermanentError, "external proxy has no endpoint")
|
||||
return ctrl.Result{}, nil
|
||||
}
|
||||
p.Status.IP = p.Spec.Endpoint.Host
|
||||
setProvisioned(p, metav1.ConditionTrue, ReasonExternalEndpoint, "tracking an external endpoint")
|
||||
r.applyHealth(p)
|
||||
return ctrl.Result{}, nil
|
||||
}
|
||||
|
||||
// providerFailure translates a classified provider error into the
|
||||
// state-machine's reaction: transient errors ride the workqueue's
|
||||
// exponential backoff, quota errors back off slowly without counting as
|
||||
// errors, and permanent errors latch Failed and stop retrying.
|
||||
func (r *ProxyReconciler) providerFailure(p *crawlv1alpha1.Proxy, err error) (ctrl.Result, error) {
|
||||
switch provider.Class(err) {
|
||||
case provider.ErrQuotaExceeded:
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonQuotaExceeded, err.Error())
|
||||
return ctrl.Result{RequeueAfter: r.QuotaRetry}, nil
|
||||
case provider.ErrPermanent:
|
||||
setProvisioned(p, metav1.ConditionFalse, ReasonPermanentError, err.Error())
|
||||
return ctrl.Result{}, nil
|
||||
default:
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
}
|
||||
|
||||
// deletionFailure is providerFailure for the finalizer path, where latching
|
||||
// a permanent failure would wedge the object forever with no retry — keep
|
||||
// retrying instead, visibly, until cleanup succeeds or an operator
|
||||
// intervenes.
|
||||
func (r *ProxyReconciler) deletionFailure(err error) (ctrl.Result, error) {
|
||||
if provider.Class(err) == provider.ErrQuotaExceeded {
|
||||
return ctrl.Result{RequeueAfter: r.QuotaRetry}, nil
|
||||
}
|
||||
return ctrl.Result{}, err
|
||||
}
|
||||
|
||||
// resolveCloudInit returns the effective cloud-init user-data, reading the
|
||||
// referenced Secret if one is used. Both the spec hash and CreateRequest see
|
||||
// only resolved content, so rotating a Secret triggers replacement.
|
||||
func (r *ProxyReconciler) resolveCloudInit(ctx context.Context, p *crawlv1alpha1.Proxy) (string, error) {
|
||||
ci := p.Spec.CloudInit
|
||||
if ci == nil {
|
||||
return "", nil
|
||||
}
|
||||
if ci.Inline != "" {
|
||||
return ci.Inline, nil
|
||||
}
|
||||
if ci.SecretRef == nil {
|
||||
return "", nil
|
||||
}
|
||||
key := ci.SecretRef.Key
|
||||
if key == "" {
|
||||
key = crawlv1alpha1.DefaultCloudInitSecretKey
|
||||
}
|
||||
var sec corev1.Secret
|
||||
if err := r.Get(ctx, client.ObjectKey{Namespace: p.Namespace, Name: ci.SecretRef.Name}, &sec); err != nil {
|
||||
return "", fmt.Errorf("resolving cloudInit secret %q: %w", ci.SecretRef.Name, err)
|
||||
}
|
||||
data, ok := sec.Data[key]
|
||||
if !ok {
|
||||
return "", fmt.Errorf("cloudInit secret %q has no key %q", ci.SecretRef.Name, key)
|
||||
}
|
||||
return string(data), nil
|
||||
}
|
||||
|
||||
// setSpecHash persists the spec-hash annotation. Status changes staged on p
|
||||
// are untouched by the Update (they live on the status subresource) and are
|
||||
// flushed by the deferred patch in Reconcile.
|
||||
func (r *ProxyReconciler) setSpecHash(ctx context.Context, p *crawlv1alpha1.Proxy, hash string) error {
|
||||
if p.Annotations[crawlv1alpha1.AnnotationSpecHash] == hash {
|
||||
return nil
|
||||
}
|
||||
if p.Annotations == nil {
|
||||
p.Annotations = map[string]string{}
|
||||
}
|
||||
p.Annotations[crawlv1alpha1.AnnotationSpecHash] = hash
|
||||
return r.Update(ctx, p)
|
||||
}
|
||||
|
||||
func placementFrom(ps *crawlv1alpha1.PlacementSpec) provider.Placement {
|
||||
if ps == nil {
|
||||
return provider.Placement{}
|
||||
}
|
||||
return provider.Placement{
|
||||
Region: ps.Region,
|
||||
Zone: ps.Zone,
|
||||
MachineType: ps.MachineType,
|
||||
Image: ps.Image,
|
||||
}
|
||||
}
|
||||
|
||||
// proxiesForSecret maps a Secret event to the Proxies whose cloudInit
|
||||
// references it, so rotating a Secret re-triggers the replacement check.
|
||||
func (r *ProxyReconciler) proxiesForSecret(ctx context.Context, obj client.Object) []reconcile.Request {
|
||||
var list crawlv1alpha1.ProxyList
|
||||
if err := r.List(ctx, &list, client.InNamespace(obj.GetNamespace())); err != nil {
|
||||
logf.FromContext(ctx).Error(err, "listing proxies for secret event", "secret", obj.GetName())
|
||||
return nil
|
||||
}
|
||||
var reqs []reconcile.Request
|
||||
for i := range list.Items {
|
||||
p := &list.Items[i]
|
||||
if ci := p.Spec.CloudInit; ci != nil && ci.SecretRef != nil && ci.SecretRef.Name == obj.GetName() {
|
||||
reqs = append(reqs, reconcile.Request{NamespacedName: client.ObjectKeyFromObject(p)})
|
||||
}
|
||||
}
|
||||
return reqs
|
||||
}
|
||||
|
||||
// SetupWithManager sets up the controller with the Manager. The Secret watch
|
||||
// only fires for Secrets the manager's cache holds; the composition root
|
||||
// (cmd/main.go) restricts that cache to labelled cloud-init Secrets.
|
||||
func (r *ProxyReconciler) SetupWithManager(mgr ctrl.Manager) error {
|
||||
return ctrl.NewControllerManagedBy(mgr).
|
||||
r.applyDefaults()
|
||||
b := ctrl.NewControllerManagedBy(mgr).
|
||||
For(&crawlv1alpha1.Proxy{}).
|
||||
Named("proxy").
|
||||
Complete(r)
|
||||
Watches(&corev1.Secret{}, handler.EnqueueRequestsFromMapFunc(r.proxiesForSecret)).
|
||||
WithOptions(controller.Options{MaxConcurrentReconciles: 3})
|
||||
if r.HealthEvents != nil {
|
||||
b = b.WatchesRawSource(source.Channel(r.HealthEvents, &handler.EnqueueRequestForObject{}))
|
||||
}
|
||||
return b.Complete(r)
|
||||
}
|
||||
|
||||
func (r *ProxyReconciler) applyDefaults() {
|
||||
if r.ProvisioningPoll == 0 {
|
||||
r.ProvisioningPoll = 10 * time.Second
|
||||
}
|
||||
if r.DriftPoll == 0 {
|
||||
r.DriftPoll = 2 * time.Minute
|
||||
}
|
||||
if r.DeletionPoll == 0 {
|
||||
r.DeletionPoll = 10 * time.Second
|
||||
}
|
||||
if r.QuotaRetry == 0 {
|
||||
r.QuotaRetry = 5 * time.Minute
|
||||
}
|
||||
if r.RequeueNow == 0 {
|
||||
r.RequeueNow = time.Second
|
||||
}
|
||||
}
|
||||
|
||||
@@ -17,77 +17,250 @@ limitations under the License.
|
||||
package controller
|
||||
|
||||
import (
|
||||
"context"
|
||||
"time"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
"k8s.io/apimachinery/pkg/api/errors"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/reconcile"
|
||||
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
var _ = Describe("Proxy Controller", func() {
|
||||
Context("When reconciling a resource", func() {
|
||||
const (
|
||||
resourceName = "test-resource"
|
||||
resourceNamespace = "default"
|
||||
)
|
||||
// These specs drive the reconciler against a real envtest API server, so
|
||||
// CRD structural defaulting and CEL validation are live — the parts the
|
||||
// fake-client action-table tests can't cover. The provider stays a stub:
|
||||
// envtest has no kubelet or cloud, so instance state is simulated by
|
||||
// mutating the stub between reconciles.
|
||||
var _ = Describe("Proxy controller", func() {
|
||||
const ns = "default"
|
||||
|
||||
ctx := context.Background()
|
||||
|
||||
typeNamespacedName := types.NamespacedName{
|
||||
Name: resourceName,
|
||||
Namespace: resourceNamespace,
|
||||
}
|
||||
proxy := &crawlv1alpha1.Proxy{}
|
||||
|
||||
BeforeEach(func() {
|
||||
By("creating the custom resource for the Kind Proxy")
|
||||
err := k8sClient.Get(ctx, typeNamespacedName, proxy)
|
||||
if err != nil && errors.IsNotFound(err) {
|
||||
resource := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: resourceName,
|
||||
Namespace: resourceNamespace,
|
||||
},
|
||||
// A minimal, schema-valid spec so this placeholder test survives the
|
||||
// CEL/CRD validation added in Step 1. Rewritten wholesale in Step 4
|
||||
// alongside the real reconciler and envtest suite.
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "10.0.0.1"},
|
||||
},
|
||||
}
|
||||
Expect(k8sClient.Create(ctx, resource)).To(Succeed())
|
||||
}
|
||||
})
|
||||
|
||||
AfterEach(func() {
|
||||
// TODO(user): Cleanup logic after each test, like removing the resource instance.
|
||||
resource := &crawlv1alpha1.Proxy{}
|
||||
err := k8sClient.Get(ctx, typeNamespacedName, resource)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
|
||||
By("Cleanup the specific resource instance Proxy")
|
||||
Expect(k8sClient.Delete(ctx, resource)).To(Succeed())
|
||||
})
|
||||
It("should successfully reconcile the resource", func() {
|
||||
By("Reconciling the created resource")
|
||||
controllerReconciler := &ProxyReconciler{
|
||||
newEnvtestReconciler := func(stub *stubProvider) *ProxyReconciler {
|
||||
return &ProxyReconciler{
|
||||
Client: k8sClient,
|
||||
Scheme: k8sClient.Scheme(),
|
||||
Providers: map[string]provider.Provider{"stub": stub},
|
||||
ProvisioningPoll: 50 * time.Millisecond,
|
||||
DriftPoll: 100 * time.Millisecond,
|
||||
DeletionPoll: 50 * time.Millisecond,
|
||||
QuotaRetry: 200 * time.Millisecond,
|
||||
RequeueNow: 10 * time.Millisecond,
|
||||
}
|
||||
}
|
||||
|
||||
_, err := controllerReconciler.Reconcile(ctx, reconcile.Request{
|
||||
NamespacedName: typeNamespacedName,
|
||||
envReconcile := func(r *ProxyReconciler, name string) (ctrl.Result, error) {
|
||||
return r.Reconcile(ctx, ctrl.Request{
|
||||
NamespacedName: types.NamespacedName{Namespace: ns, Name: name},
|
||||
})
|
||||
}
|
||||
|
||||
fetch := func(name string) *crawlv1alpha1.Proxy {
|
||||
p := &crawlv1alpha1.Proxy{}
|
||||
Expect(k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, p)).To(Succeed())
|
||||
return p
|
||||
}
|
||||
|
||||
// cleanup drives a Managed proxy's finalizer to completion so one spec's
|
||||
// leftovers can't leak into another. Registered via DeferCleanup so it
|
||||
// runs even when the spec body fails mid-way.
|
||||
cleanup := func(r *ProxyReconciler, stub *stubProvider, name string) {
|
||||
p := &crawlv1alpha1.Proxy{}
|
||||
err := k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, p)
|
||||
if apierrors.IsNotFound(err) {
|
||||
return
|
||||
}
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
// TODO(user): Add more specific assertions depending on your controller's reconciliation logic.
|
||||
// Example: If you expect a certain status condition after reconciliation, verify it here.
|
||||
})
|
||||
Expect(k8sClient.Delete(ctx, p)).To(Succeed())
|
||||
stub.getErr = provider.Wrap(provider.ErrNotFound, "get", "stub", "", nil)
|
||||
for range 3 {
|
||||
if _, err := envReconcile(r, name); err != nil {
|
||||
break
|
||||
}
|
||||
if apierrors.IsNotFound(k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, p)) {
|
||||
return
|
||||
}
|
||||
}
|
||||
Fail("cleanup did not drive the proxy " + name + " to deletion")
|
||||
}
|
||||
|
||||
managedSpec := func() crawlv1alpha1.ProxySpec {
|
||||
return crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged,
|
||||
Provider: "stub",
|
||||
}
|
||||
}
|
||||
|
||||
It("provisions a Managed proxy through to Running", func() {
|
||||
const name = "e2e-provision"
|
||||
stub := &stubProvider{createID: "inst-1"}
|
||||
r := newEnvtestReconciler(stub)
|
||||
DeferCleanup(func() { cleanup(r, stub, name) })
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: managedSpec(),
|
||||
})).To(Succeed())
|
||||
|
||||
By("adding the finalizer on the first pass")
|
||||
res, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res).To(Equal(ctrl.Result{}))
|
||||
Expect(fetch(name).Finalizers).To(ContainElement(crawlv1alpha1.FinalizerName))
|
||||
|
||||
By("creating the instance on the second pass")
|
||||
res, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.ProvisioningPoll))
|
||||
p := fetch(name)
|
||||
Expect(p.Status.ProviderID).To(Equal("inst-1"))
|
||||
Expect(p.Annotations).To(HaveKey(crawlv1alpha1.AnnotationSpecHash))
|
||||
// The real API server defaulted spec.port; the create request must
|
||||
// have seen it.
|
||||
Expect(p.Spec.Port).To(Equal(crawlv1alpha1.DefaultPort))
|
||||
Expect(stub.lastCreate.Port).To(Equal(crawlv1alpha1.DefaultPort))
|
||||
Expect(stub.lastCreate.Name).To(Equal(provider.NameFromUID(p.UID)))
|
||||
|
||||
By("polling while the instance provisions")
|
||||
stub.getInst = &provider.Instance{ID: "inst-1", State: provider.StateProvisioning}
|
||||
res, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.ProvisioningPoll))
|
||||
Expect(fetch(name).Status.Phase).To(Equal(crawlv1alpha1.PhaseProvisioning))
|
||||
|
||||
By("publishing the IP once the instance runs")
|
||||
stub.getInst = &provider.Instance{ID: "inst-1", IP: "10.9.8.7", State: provider.StateRunning}
|
||||
res, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.DriftPoll))
|
||||
p = fetch(name)
|
||||
Expect(p.Status.IP).To(Equal("10.9.8.7"))
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned)
|
||||
Expect(cond).NotTo(BeNil())
|
||||
Expect(cond.Status).To(Equal(metav1.ConditionTrue))
|
||||
Expect(p.Status.ObservedGeneration).To(Equal(p.Generation))
|
||||
})
|
||||
|
||||
It("replaces the instance when the spec changes", func() {
|
||||
const name = "e2e-replace"
|
||||
stub := &stubProvider{createID: "inst-old"}
|
||||
r := newEnvtestReconciler(stub)
|
||||
DeferCleanup(func() { cleanup(r, stub, name) })
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: managedSpec(),
|
||||
})).To(Succeed())
|
||||
_, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
stub.getInst = &provider.Instance{ID: "inst-old", IP: "10.0.0.1", State: provider.StateRunning}
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
oldHash := fetch(name).Annotations[crawlv1alpha1.AnnotationSpecHash]
|
||||
|
||||
By("editing a replacement-triggering field")
|
||||
p := fetch(name)
|
||||
p.Spec.Port = 8080
|
||||
Expect(k8sClient.Update(ctx, p)).To(Succeed())
|
||||
|
||||
By("deleting the old instance first")
|
||||
res, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.DeletionPoll))
|
||||
Expect(stub.deleteCalls).To(Equal(1))
|
||||
p = fetch(name)
|
||||
Expect(p.Annotations[crawlv1alpha1.AnnotationSpecHash]).To(Equal(oldHash),
|
||||
"hash must not advance while the old instance still exists")
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned)
|
||||
Expect(cond.Reason).To(Equal(ReasonReplacing))
|
||||
|
||||
By("advancing the hash once the old instance is gone")
|
||||
stub.getErr = provider.Wrap(provider.ErrNotFound, "get", "stub", "inst-old", nil)
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
p = fetch(name)
|
||||
Expect(p.Status.ProviderID).To(BeEmpty())
|
||||
Expect(p.Annotations[crawlv1alpha1.AnnotationSpecHash]).NotTo(Equal(oldHash))
|
||||
|
||||
By("creating the replacement")
|
||||
stub.createID = "inst-new"
|
||||
stub.getErr = nil
|
||||
stub.getInst = nil
|
||||
res, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.ProvisioningPoll))
|
||||
Expect(stub.createCalls).To(Equal(2))
|
||||
Expect(fetch(name).Status.ProviderID).To(Equal("inst-new"))
|
||||
Expect(stub.lastCreate.Port).To(Equal(int32(8080)))
|
||||
})
|
||||
|
||||
It("cleans up the instance on delete via the finalizer", func() {
|
||||
const name = "e2e-delete"
|
||||
stub := &stubProvider{createID: "inst-del"}
|
||||
r := newEnvtestReconciler(stub)
|
||||
DeferCleanup(func() { cleanup(r, stub, name) })
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: managedSpec(),
|
||||
})).To(Succeed())
|
||||
_, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
stub.getInst = &provider.Instance{ID: "inst-del", IP: "10.0.0.2", State: provider.StateRunning}
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
|
||||
By("deleting the CR — the finalizer holds it")
|
||||
Expect(k8sClient.Delete(ctx, fetch(name))).To(Succeed())
|
||||
res, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res.RequeueAfter).To(Equal(r.DeletionPoll))
|
||||
Expect(stub.deleteCalls).To(Equal(1))
|
||||
Expect(fetch(name).Status.Phase).To(Equal(crawlv1alpha1.PhaseDeleting))
|
||||
|
||||
By("removing the finalizer once the instance is gone")
|
||||
stub.getErr = provider.Wrap(provider.ErrNotFound, "get", "stub", "inst-del", nil)
|
||||
_, err = envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
err = k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &crawlv1alpha1.Proxy{})
|
||||
Expect(apierrors.IsNotFound(err)).To(BeTrue(), "proxy should be fully deleted")
|
||||
})
|
||||
|
||||
It("tracks an External proxy without touching providers", func() {
|
||||
const name = "e2e-external"
|
||||
stub := &stubProvider{}
|
||||
r := newEnvtestReconciler(stub)
|
||||
|
||||
Expect(k8sClient.Create(ctx, &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: ns},
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "203.0.113.7"},
|
||||
},
|
||||
})).To(Succeed())
|
||||
|
||||
res, err := envReconcile(r, name)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(res).To(Equal(ctrl.Result{}))
|
||||
|
||||
p := fetch(name)
|
||||
Expect(p.Status.IP).To(Equal("203.0.113.7"))
|
||||
Expect(p.Finalizers).To(BeEmpty())
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned)
|
||||
Expect(cond).NotTo(BeNil())
|
||||
Expect(cond.Reason).To(Equal(ReasonExternalEndpoint))
|
||||
Expect(stub.createCalls + stub.getCalls + stub.deleteCalls).To(BeZero())
|
||||
|
||||
By("deleting without any finalizer round-trip")
|
||||
Expect(k8sClient.Delete(ctx, p)).To(Succeed())
|
||||
err = k8sClient.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &crawlv1alpha1.Proxy{})
|
||||
Expect(apierrors.IsNotFound(err)).To(BeTrue())
|
||||
})
|
||||
})
|
||||
|
||||
550
internal/controller/reconcile_test.go
Normal file
550
internal/controller/reconcile_test.go
Normal file
@@ -0,0 +1,550 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// Distinct per-interval values so an asserted ctrl.Result is unambiguous
|
||||
// about which action-table row produced it.
|
||||
const (
|
||||
tProvisioningPoll = 11 * time.Second
|
||||
tDriftPoll = 22 * time.Second
|
||||
tDeletionPoll = 33 * time.Second
|
||||
tQuotaRetry = 44 * time.Second
|
||||
tRequeueNow = 55 * time.Millisecond
|
||||
)
|
||||
|
||||
const (
|
||||
testProxyName = "p1"
|
||||
testNamespace = "default"
|
||||
testUID = types.UID("11111111-2222-3333-4444-555555555555")
|
||||
)
|
||||
|
||||
// stubProvider is the plan's in-test Provider stub: a handful of lines, no
|
||||
// config format, no fault-injection surface beyond settable fields.
|
||||
type stubProvider struct {
|
||||
createID string
|
||||
createErr error
|
||||
getInst *provider.Instance
|
||||
getErr error
|
||||
deleteErr error
|
||||
|
||||
createCalls, getCalls, deleteCalls int
|
||||
lastCreate provider.CreateRequest
|
||||
}
|
||||
|
||||
func (s *stubProvider) Create(_ context.Context, req provider.CreateRequest) (string, error) {
|
||||
s.createCalls++
|
||||
s.lastCreate = req
|
||||
if s.createErr != nil {
|
||||
return "", s.createErr
|
||||
}
|
||||
return s.createID, nil
|
||||
}
|
||||
|
||||
func (s *stubProvider) Get(_ context.Context, _ string) (*provider.Instance, error) {
|
||||
s.getCalls++
|
||||
if s.getErr != nil {
|
||||
return nil, s.getErr
|
||||
}
|
||||
return s.getInst, nil
|
||||
}
|
||||
|
||||
func (s *stubProvider) Delete(_ context.Context, _ string) error {
|
||||
s.deleteCalls++
|
||||
return s.deleteErr
|
||||
}
|
||||
|
||||
func (s *stubProvider) ListByTag(context.Context) ([]provider.Instance, error) {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
func notFoundErr() error {
|
||||
return provider.Wrap(provider.ErrNotFound, "get", "stub", "some-id", nil)
|
||||
}
|
||||
|
||||
func testScheme(t *testing.T) *runtime.Scheme {
|
||||
t.Helper()
|
||||
s := runtime.NewScheme()
|
||||
if err := crawlv1alpha1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("adding crawl scheme: %v", err)
|
||||
}
|
||||
if err := corev1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("adding core scheme: %v", err)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// managedProxy returns a Managed proxy that already carries the finalizer —
|
||||
// the state most action-table rows start from. Mutators adjust from there.
|
||||
func managedProxy(mut ...func(*crawlv1alpha1.Proxy)) *crawlv1alpha1.Proxy {
|
||||
p := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: testProxyName,
|
||||
Namespace: testNamespace,
|
||||
UID: testUID,
|
||||
Generation: 1,
|
||||
Finalizers: []string{crawlv1alpha1.FinalizerName},
|
||||
},
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged,
|
||||
Provider: "stub",
|
||||
},
|
||||
}
|
||||
for _, m := range mut {
|
||||
m(p)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
func withProviderID(id string) func(*crawlv1alpha1.Proxy) {
|
||||
return func(p *crawlv1alpha1.Proxy) { p.Status.ProviderID = id }
|
||||
}
|
||||
|
||||
func withSpecHashAnnotation(hash string) func(*crawlv1alpha1.Proxy) {
|
||||
return func(p *crawlv1alpha1.Proxy) {
|
||||
if p.Annotations == nil {
|
||||
p.Annotations = map[string]string{}
|
||||
}
|
||||
p.Annotations[crawlv1alpha1.AnnotationSpecHash] = hash
|
||||
}
|
||||
}
|
||||
|
||||
func deleting() func(*crawlv1alpha1.Proxy) {
|
||||
return func(p *crawlv1alpha1.Proxy) {
|
||||
now := metav1.Now()
|
||||
p.DeletionTimestamp = &now
|
||||
}
|
||||
}
|
||||
|
||||
func newTestReconciler(t *testing.T, stub *stubProvider, objs ...client.Object) *ProxyReconciler {
|
||||
t.Helper()
|
||||
c := fake.NewClientBuilder().
|
||||
WithScheme(testScheme(t)).
|
||||
WithStatusSubresource(&crawlv1alpha1.Proxy{}).
|
||||
WithObjects(objs...).
|
||||
Build()
|
||||
return &ProxyReconciler{
|
||||
Client: c,
|
||||
Providers: map[string]provider.Provider{"stub": stub},
|
||||
ProvisioningPoll: tProvisioningPoll,
|
||||
DriftPoll: tDriftPoll,
|
||||
DeletionPoll: tDeletionPoll,
|
||||
QuotaRetry: tQuotaRetry,
|
||||
RequeueNow: tRequeueNow,
|
||||
}
|
||||
}
|
||||
|
||||
func doReconcile(t *testing.T, r *ProxyReconciler) (ctrl.Result, error) {
|
||||
t.Helper()
|
||||
return r.Reconcile(context.Background(), ctrl.Request{
|
||||
NamespacedName: types.NamespacedName{Namespace: testNamespace, Name: testProxyName},
|
||||
})
|
||||
}
|
||||
|
||||
func getProxy(t *testing.T, r *ProxyReconciler) *crawlv1alpha1.Proxy {
|
||||
t.Helper()
|
||||
var p crawlv1alpha1.Proxy
|
||||
if err := r.Get(context.Background(), types.NamespacedName{Namespace: testNamespace, Name: testProxyName}, &p); err != nil {
|
||||
t.Fatalf("getting proxy: %v", err)
|
||||
}
|
||||
return &p
|
||||
}
|
||||
|
||||
func assertCondition(t *testing.T, p *crawlv1alpha1.Proxy, condType string, status metav1.ConditionStatus, reason string) {
|
||||
t.Helper()
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, condType)
|
||||
if cond == nil {
|
||||
t.Fatalf("condition %s missing, have %+v", condType, p.Status.Conditions)
|
||||
}
|
||||
if cond.Status != status || cond.Reason != reason {
|
||||
t.Errorf("condition %s = %s/%s, want %s/%s", condType, cond.Status, cond.Reason, status, reason)
|
||||
}
|
||||
if cond.ObservedGeneration != p.Generation {
|
||||
t.Errorf("condition %s observedGeneration = %d, want %d", condType, cond.ObservedGeneration, p.Generation)
|
||||
}
|
||||
}
|
||||
|
||||
// TestReconcile_actionTable exercises every row of the plan's action table
|
||||
// by calling Reconcile directly against a fake client. Caveat (documented in
|
||||
// the plan): the fake client runs neither CEL validation nor structural
|
||||
// defaulting — the envtest suite covers those.
|
||||
func TestReconcile_actionTable(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// The fixture's hash: no placement, no cloud-init, defaulted port.
|
||||
freshHash := specHash(managedProxy(), "")
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
proxy *crawlv1alpha1.Proxy
|
||||
extraObjs []client.Object
|
||||
stub *stubProvider
|
||||
wantResult ctrl.Result
|
||||
wantErr bool
|
||||
verify func(t *testing.T, r *ProxyReconciler, stub *stubProvider)
|
||||
}{
|
||||
{
|
||||
name: "managed without finalizer gets one and stops",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Finalizers = nil
|
||||
}),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if len(p.Finalizers) != 1 || p.Finalizers[0] != crawlv1alpha1.FinalizerName {
|
||||
t.Errorf("finalizers = %v, want [%s]", p.Finalizers, crawlv1alpha1.FinalizerName)
|
||||
}
|
||||
if stub.createCalls+stub.getCalls+stub.deleteCalls != 0 {
|
||||
t.Errorf("provider was called before the finalizer was persisted")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "empty providerID creates the instance",
|
||||
proxy: managedProxy(),
|
||||
stub: &stubProvider{createID: "stub-id-1"},
|
||||
wantResult: ctrl.Result{RequeueAfter: tProvisioningPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.ProviderID != "stub-id-1" {
|
||||
t.Errorf("providerID = %q, want stub-id-1", p.Status.ProviderID)
|
||||
}
|
||||
if got := p.Annotations[crawlv1alpha1.AnnotationSpecHash]; got != freshHash {
|
||||
t.Errorf("spec-hash annotation = %q, want %q", got, freshHash)
|
||||
}
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonProvisioning)
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseProvisioning {
|
||||
t.Errorf("phase = %s, want Provisioning", p.Status.Phase)
|
||||
}
|
||||
want := provider.CreateRequest{
|
||||
Name: provider.NameFromUID(testUID),
|
||||
UID: string(testUID),
|
||||
Namespace: testNamespace,
|
||||
ProxyName: testProxyName,
|
||||
Port: crawlv1alpha1.DefaultPort,
|
||||
}
|
||||
if stub.lastCreate != want {
|
||||
t.Errorf("CreateRequest = %+v, want %+v", stub.lastCreate, want)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "provisioning instance polls again",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: &stubProvider{getInst: &provider.Instance{ID: "stub-id-1", State: provider.StateProvisioning}},
|
||||
wantResult: ctrl.Result{RequeueAfter: tProvisioningPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.IP != "" {
|
||||
t.Errorf("ip = %q, want empty while provisioning", p.Status.IP)
|
||||
}
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonProvisioning)
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "running instance publishes IP",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: &stubProvider{
|
||||
getInst: &provider.Instance{ID: "stub-id-1", IP: "10.1.2.3", State: provider.StateRunning},
|
||||
},
|
||||
wantResult: ctrl.Result{RequeueAfter: tDriftPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.IP != "10.1.2.3" {
|
||||
t.Errorf("ip = %q, want 10.1.2.3", p.Status.IP)
|
||||
}
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated)
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseProvisioning {
|
||||
t.Errorf("phase = %s, want Provisioning until a health verdict exists", p.Status.Phase)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "stopped instance is deleted for recreation",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash)),
|
||||
stub: &stubProvider{getInst: &provider.Instance{ID: "stub-id-1", State: provider.StateStopped}},
|
||||
wantResult: ctrl.Result{RequeueAfter: tDeletionPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
if stub.deleteCalls != 1 {
|
||||
t.Errorf("deleteCalls = %d, want 1", stub.deleteCalls)
|
||||
}
|
||||
assertCondition(t, getProxy(t, r), crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonRecreating)
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "vanished instance clears ID for recreation",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation(freshHash),
|
||||
func(p *crawlv1alpha1.Proxy) { p.Status.IP = "10.1.2.3" }),
|
||||
stub: &stubProvider{getErr: notFoundErr()},
|
||||
wantResult: ctrl.Result{RequeueAfter: tRequeueNow},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.ProviderID != "" || p.Status.IP != "" {
|
||||
t.Errorf("providerID/ip = %q/%q, want both cleared", p.Status.ProviderID, p.Status.IP)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "hash mismatch deletes the old instance",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation("stale-hash")),
|
||||
stub: &stubProvider{
|
||||
getInst: &provider.Instance{ID: "stub-id-1", IP: "10.1.2.3", State: provider.StateRunning},
|
||||
},
|
||||
wantResult: ctrl.Result{RequeueAfter: tDeletionPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
if stub.deleteCalls != 1 {
|
||||
t.Errorf("deleteCalls = %d, want 1", stub.deleteCalls)
|
||||
}
|
||||
p := getProxy(t, r)
|
||||
// The hash must not advance until the old instance is gone,
|
||||
// or a crash would strand a half-replaced proxy.
|
||||
if got := p.Annotations[crawlv1alpha1.AnnotationSpecHash]; got != "stale-hash" {
|
||||
t.Errorf("spec-hash annotation = %q, want still stale-hash", got)
|
||||
}
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonReplacing)
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "empty annotation adopts instead of replacing",
|
||||
proxy: managedProxy(withProviderID("stub-id-1")),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{RequeueAfter: tRequeueNow},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if got := p.Annotations[crawlv1alpha1.AnnotationSpecHash]; got != freshHash {
|
||||
t.Errorf("spec-hash annotation = %q, want %q", got, freshHash)
|
||||
}
|
||||
if p.Status.ProviderID != "stub-id-1" {
|
||||
t.Errorf("providerID = %q, want untouched stub-id-1", p.Status.ProviderID)
|
||||
}
|
||||
if stub.deleteCalls != 0 {
|
||||
t.Errorf("deleteCalls = %d, want 0 — adoption must not replace", stub.deleteCalls)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "mismatch with instance gone advances the hash",
|
||||
proxy: managedProxy(withProviderID("stub-id-1"), withSpecHashAnnotation("stale-hash")),
|
||||
stub: &stubProvider{getErr: notFoundErr()},
|
||||
wantResult: ctrl.Result{RequeueAfter: tRequeueNow},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.ProviderID != "" {
|
||||
t.Errorf("providerID = %q, want cleared", p.Status.ProviderID)
|
||||
}
|
||||
if got := p.Annotations[crawlv1alpha1.AnnotationSpecHash]; got != freshHash {
|
||||
t.Errorf("spec-hash annotation = %q, want advanced to %q", got, freshHash)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "deletion with no providerID removes the finalizer",
|
||||
proxy: managedProxy(deleting()),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
assertProxyGone(t, r)
|
||||
if stub.deleteCalls != 0 {
|
||||
t.Errorf("deleteCalls = %d, want 0", stub.deleteCalls)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "deletion with instance already gone removes the finalizer",
|
||||
proxy: managedProxy(deleting(), withProviderID("stub-id-1")),
|
||||
stub: &stubProvider{getErr: notFoundErr()},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
assertProxyGone(t, r)
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "deletion deletes the instance and polls",
|
||||
proxy: managedProxy(deleting(), withProviderID("stub-id-1")),
|
||||
stub: &stubProvider{
|
||||
getInst: &provider.Instance{ID: "stub-id-1", State: provider.StateRunning},
|
||||
},
|
||||
wantResult: ctrl.Result{RequeueAfter: tDeletionPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
if stub.deleteCalls != 1 {
|
||||
t.Errorf("deleteCalls = %d, want 1", stub.deleteCalls)
|
||||
}
|
||||
p := getProxy(t, r)
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseDeleting {
|
||||
t.Errorf("phase = %s, want Deleting", p.Status.Phase)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "external proxy tracks its endpoint without a finalizer",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Finalizers = nil
|
||||
p.Spec = crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: "203.0.113.7", Port: 8080},
|
||||
}
|
||||
}),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.IP != "203.0.113.7" {
|
||||
t.Errorf("ip = %q, want the endpoint host", p.Status.IP)
|
||||
}
|
||||
if len(p.Finalizers) != 0 {
|
||||
t.Errorf("finalizers = %v, want none on External", p.Finalizers)
|
||||
}
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonExternalEndpoint)
|
||||
if stub.createCalls+stub.getCalls+stub.deleteCalls != 0 {
|
||||
t.Errorf("provider was called for an External proxy")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "quota error backs off slowly without failing",
|
||||
proxy: managedProxy(),
|
||||
stub: &stubProvider{
|
||||
createErr: provider.Wrap(provider.ErrQuotaExceeded, "create", "stub", "", nil),
|
||||
},
|
||||
wantResult: ctrl.Result{RequeueAfter: tQuotaRetry},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonQuotaExceeded)
|
||||
if p.Status.Phase == crawlv1alpha1.PhaseFailed {
|
||||
t.Errorf("phase = Failed, want anything but — quota is a wait, not a failure")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "permanent error latches Failed and stops calling the provider",
|
||||
proxy: managedProxy(),
|
||||
stub: &stubProvider{
|
||||
createErr: provider.Wrap(provider.ErrPermanent, "create", "stub", "", nil),
|
||||
},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonPermanentError)
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseFailed {
|
||||
t.Errorf("phase = %s, want Failed", p.Status.Phase)
|
||||
}
|
||||
if res, err := doReconcile(t, r); err != nil || res != (ctrl.Result{}) {
|
||||
t.Errorf("second reconcile = %+v, %v; want empty result, nil", res, err)
|
||||
}
|
||||
if stub.createCalls != 1 {
|
||||
t.Errorf("createCalls = %d after latch, want 1", stub.createCalls)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "transient error is returned for workqueue backoff",
|
||||
proxy: managedProxy(),
|
||||
stub: &stubProvider{
|
||||
createErr: provider.Wrap(provider.ErrTransient, "create", "stub", "", nil),
|
||||
},
|
||||
wantResult: ctrl.Result{},
|
||||
wantErr: true,
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
if p.Status.Phase != crawlv1alpha1.PhasePending {
|
||||
t.Errorf("phase = %s, want still Pending", p.Status.Phase)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "unconfigured provider is a permanent failure",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Spec.Provider = "no-such-provider"
|
||||
}),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, _ *stubProvider) {
|
||||
p := getProxy(t, r)
|
||||
assertCondition(t, p, crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonPermanentError)
|
||||
if p.Status.Phase != crawlv1alpha1.PhaseFailed {
|
||||
t.Errorf("phase = %s, want Failed", p.Status.Phase)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "cloud-init secret is resolved into the create request",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Spec.CloudInit = &crawlv1alpha1.CloudInitSpec{
|
||||
SecretRef: &crawlv1alpha1.SecretKeySelector{Name: "ci-secret"},
|
||||
}
|
||||
}),
|
||||
extraObjs: []client.Object{
|
||||
&corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "ci-secret", Namespace: testNamespace},
|
||||
Data: map[string][]byte{"user-data": []byte("#cloud-config\npackages: [squid]")},
|
||||
},
|
||||
},
|
||||
stub: &stubProvider{createID: "stub-id-1"},
|
||||
wantResult: ctrl.Result{RequeueAfter: tProvisioningPoll},
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
if want := "#cloud-config\npackages: [squid]"; stub.lastCreate.CloudInit != want {
|
||||
t.Errorf("CreateRequest.CloudInit = %q, want the resolved secret content", stub.lastCreate.CloudInit)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "missing cloud-init secret errors and marks the condition",
|
||||
proxy: managedProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Spec.CloudInit = &crawlv1alpha1.CloudInitSpec{
|
||||
SecretRef: &crawlv1alpha1.SecretKeySelector{Name: "absent"},
|
||||
}
|
||||
}),
|
||||
stub: &stubProvider{},
|
||||
wantResult: ctrl.Result{},
|
||||
wantErr: true,
|
||||
verify: func(t *testing.T, r *ProxyReconciler, stub *stubProvider) {
|
||||
assertCondition(t, getProxy(t, r), crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonCloudInitError)
|
||||
if stub.createCalls != 0 {
|
||||
t.Errorf("createCalls = %d, want 0 with unresolved cloud-init", stub.createCalls)
|
||||
}
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
objs := append([]client.Object{tc.proxy}, tc.extraObjs...)
|
||||
r := newTestReconciler(t, tc.stub, objs...)
|
||||
res, err := doReconcile(t, r)
|
||||
if (err != nil) != tc.wantErr {
|
||||
t.Fatalf("Reconcile error = %v, wantErr %v", err, tc.wantErr)
|
||||
}
|
||||
if res != tc.wantResult {
|
||||
t.Errorf("Result = %+v, want %+v", res, tc.wantResult)
|
||||
}
|
||||
tc.verify(t, r, tc.stub)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func assertProxyGone(t *testing.T, r *ProxyReconciler) {
|
||||
t.Helper()
|
||||
var p crawlv1alpha1.Proxy
|
||||
err := r.Get(context.Background(), types.NamespacedName{Namespace: testNamespace, Name: testProxyName}, &p)
|
||||
if !apierrors.IsNotFound(err) {
|
||||
t.Errorf("proxy still exists (err=%v), want NotFound after finalizer removal", err)
|
||||
}
|
||||
}
|
||||
41
internal/controller/spechash.go
Normal file
41
internal/controller/spechash.go
Normal file
@@ -0,0 +1,41 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
// specHashInput is the explicit set of fields whose change requires
|
||||
// replacing the instance. Deliberately not ProxySpec wholesale: adding a
|
||||
// spec field that doesn't affect the VM (attributes, maxLeases, healthCheck)
|
||||
// must not churn the fleet on operator upgrade.
|
||||
type specHashInput struct {
|
||||
Placement *crawlv1alpha1.PlacementSpec `json:"placement,omitempty"`
|
||||
CloudInit string `json:"cloudInit,omitempty"`
|
||||
Port int32 `json:"port"`
|
||||
}
|
||||
|
||||
// specHash returns the hex SHA-256 of the replacement-triggering spec
|
||||
// fields. cloudInit is the already-resolved content, so rotating a
|
||||
// referenced Secret changes the hash even though the spec is untouched.
|
||||
func specHash(p *crawlv1alpha1.Proxy, cloudInit string) string {
|
||||
in := specHashInput{
|
||||
Placement: p.Spec.Placement,
|
||||
CloudInit: cloudInit,
|
||||
Port: p.EffectivePort(),
|
||||
}
|
||||
// nil and empty placement mean the same thing; hash them identically.
|
||||
if in.Placement != nil && *in.Placement == (crawlv1alpha1.PlacementSpec{}) {
|
||||
in.Placement = nil
|
||||
}
|
||||
b, err := json.Marshal(in)
|
||||
if err != nil {
|
||||
// A struct of strings and an int32 cannot fail to marshal.
|
||||
panic(err)
|
||||
}
|
||||
sum := sha256.Sum256(b)
|
||||
return hex.EncodeToString(sum[:])
|
||||
}
|
||||
118
internal/controller/spechash_test.go
Normal file
118
internal/controller/spechash_test.go
Normal file
@@ -0,0 +1,118 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"testing"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
func hashProxy(mut ...func(*crawlv1alpha1.Proxy)) *crawlv1alpha1.Proxy {
|
||||
p := &crawlv1alpha1.Proxy{
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeManaged,
|
||||
Provider: "stub",
|
||||
Port: 3128,
|
||||
Placement: &crawlv1alpha1.PlacementSpec{
|
||||
Zone: "europe-west1-b",
|
||||
MachineType: "e2-micro",
|
||||
},
|
||||
},
|
||||
}
|
||||
for _, m := range mut {
|
||||
m(p)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
func TestSpecHash_stability(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
p := hashProxy()
|
||||
h1 := specHash(p, "cloud-init-content")
|
||||
h2 := specHash(p.DeepCopy(), "cloud-init-content")
|
||||
if h1 != h2 {
|
||||
t.Errorf("same input hashed differently: %s vs %s", h1, h2)
|
||||
}
|
||||
if !regexp.MustCompile(`^[0-9a-f]{64}$`).MatchString(h1) {
|
||||
t.Errorf("hash %q is not hex SHA-256", h1)
|
||||
}
|
||||
|
||||
// Fields outside the replacement set must not affect the hash — that is
|
||||
// the whole point of an explicit hash-input struct.
|
||||
q := hashProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Spec.Attributes = map[string]string{"geo": "eu"}
|
||||
five := int32(5)
|
||||
p.Spec.MaxLeases = &five
|
||||
p.Spec.HealthCheck = &crawlv1alpha1.HealthCheckSpec{IntervalSeconds: 60}
|
||||
})
|
||||
if specHash(q, "cloud-init-content") != h1 {
|
||||
t.Error("non-replacement spec fields changed the hash")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSpecHash_normalization(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
nilPlacement := hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Placement = nil })
|
||||
emptyPlacement := hashProxy(func(p *crawlv1alpha1.Proxy) {
|
||||
p.Spec.Placement = &crawlv1alpha1.PlacementSpec{}
|
||||
})
|
||||
if specHash(nilPlacement, "") != specHash(emptyPlacement, "") {
|
||||
t.Error("nil and empty placement hashed differently")
|
||||
}
|
||||
|
||||
// An unset port and an explicit default port mean the same instance.
|
||||
unsetPort := hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Port = 0 })
|
||||
defaultPort := hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Port = crawlv1alpha1.DefaultPort })
|
||||
if specHash(unsetPort, "") != specHash(defaultPort, "") {
|
||||
t.Error("unset port and explicit default port hashed differently")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSpecHash_sensitivity(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
base := specHash(hashProxy(), "cloud-init")
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
proxy *crawlv1alpha1.Proxy
|
||||
cloudInit string
|
||||
}{
|
||||
{
|
||||
name: "port change",
|
||||
proxy: hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Port = 8080 }),
|
||||
cloudInit: "cloud-init",
|
||||
},
|
||||
{
|
||||
name: "zone change",
|
||||
proxy: hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Placement.Zone = "us-east1-c" }),
|
||||
cloudInit: "cloud-init",
|
||||
},
|
||||
{
|
||||
name: "machine type change",
|
||||
proxy: hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Placement.MachineType = "e2-small" }),
|
||||
cloudInit: "cloud-init",
|
||||
},
|
||||
{
|
||||
name: "image change",
|
||||
proxy: hashProxy(func(p *crawlv1alpha1.Proxy) { p.Spec.Placement.Image = "debian-13" }),
|
||||
cloudInit: "cloud-init",
|
||||
},
|
||||
{
|
||||
name: "resolved cloud-init change (secret rotation)",
|
||||
proxy: hashProxy(),
|
||||
cloudInit: "rotated-cloud-init",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
if specHash(tc.proxy, tc.cloudInit) == base {
|
||||
t.Error("hash did not change")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
121
internal/controller/status.go
Normal file
121
internal/controller/status.go
Normal file
@@ -0,0 +1,121 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"k8s.io/apimachinery/pkg/api/equality"
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
// Reasons used on the Provisioned condition, plus the two the reconciler
|
||||
// writes on the Healthy condition when representing the health engine's
|
||||
// verdict (the engine owns the state; only the reconciler writes status).
|
||||
const (
|
||||
ReasonProvisioning = "Provisioning"
|
||||
ReasonCreated = "Created"
|
||||
ReasonReplacing = "Replacing"
|
||||
ReasonRecreating = "Recreating"
|
||||
ReasonQuotaExceeded = "QuotaExceeded"
|
||||
ReasonPermanentError = "PermanentError"
|
||||
ReasonCloudInitError = "CloudInitError"
|
||||
ReasonExternalEndpoint = "ExternalEndpoint"
|
||||
ReasonDeleting = "Deleting"
|
||||
|
||||
ReasonProbeSucceeded = "ProbeSucceeded"
|
||||
ReasonProbeFailed = "ProbeFailed"
|
||||
)
|
||||
|
||||
// setProvisioned stages the Provisioned condition on p. Nothing is written
|
||||
// to the API server here; the deferred patch in Reconcile flushes it.
|
||||
// ObservedGeneration is passed explicitly — SetStatusCondition does not
|
||||
// populate it, and without it every condition would report generation 0.
|
||||
func setProvisioned(p *crawlv1alpha1.Proxy, status metav1.ConditionStatus, reason, message string) {
|
||||
apimeta.SetStatusCondition(&p.Status.Conditions, metav1.Condition{
|
||||
Type: crawlv1alpha1.ConditionProvisioned,
|
||||
Status: status,
|
||||
Reason: reason,
|
||||
Message: message,
|
||||
ObservedGeneration: p.Generation,
|
||||
})
|
||||
}
|
||||
|
||||
// applyHealth stages the Healthy condition and the latency fields from the
|
||||
// health engine's current snapshot. Called only from states where the proxy
|
||||
// is reachable (Running, External); everywhere else the condition is either
|
||||
// left as-is or removed by the create branch.
|
||||
func (r *ProxyReconciler) applyHealth(p *crawlv1alpha1.Proxy) {
|
||||
if r.Health == nil {
|
||||
return
|
||||
}
|
||||
snap, ok := r.Health.Snapshot(client.ObjectKeyFromObject(p))
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
cond := metav1.Condition{
|
||||
Type: crawlv1alpha1.ConditionHealthy,
|
||||
ObservedGeneration: p.Generation,
|
||||
}
|
||||
if snap.Healthy {
|
||||
cond.Status = metav1.ConditionTrue
|
||||
cond.Reason = ReasonProbeSucceeded
|
||||
cond.Message = "probe succeeded through the proxy"
|
||||
} else {
|
||||
cond.Status = metav1.ConditionFalse
|
||||
cond.Reason = ReasonProbeFailed
|
||||
cond.Message = fmt.Sprintf("%d consecutive probe failures; last: %s",
|
||||
snap.ConsecutiveFailures, snap.LastError)
|
||||
}
|
||||
apimeta.SetStatusCondition(&p.Status.Conditions, cond)
|
||||
p.Status.LatencyMillis = snap.Latency.Milliseconds()
|
||||
if !snap.LastProbe.IsZero() {
|
||||
p.Status.LastHealthCheckTime = &metav1.Time{Time: snap.LastProbe}
|
||||
}
|
||||
}
|
||||
|
||||
// computePhase derives status.phase from deletionTimestamp and the
|
||||
// Provisioned/Healthy conditions. Pure, so the truth table is unit-testable.
|
||||
func computePhase(p *crawlv1alpha1.Proxy) crawlv1alpha1.ProxyPhase {
|
||||
if !p.DeletionTimestamp.IsZero() {
|
||||
return crawlv1alpha1.PhaseDeleting
|
||||
}
|
||||
prov := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionProvisioned)
|
||||
if prov == nil {
|
||||
return crawlv1alpha1.PhasePending
|
||||
}
|
||||
if prov.Status != metav1.ConditionTrue {
|
||||
// Quota exhaustion is a slow-retry wait, not a terminal state — only
|
||||
// a permanent error latches Failed.
|
||||
if prov.Reason == ReasonPermanentError {
|
||||
return crawlv1alpha1.PhaseFailed
|
||||
}
|
||||
return crawlv1alpha1.PhaseProvisioning
|
||||
}
|
||||
healthy := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy)
|
||||
switch {
|
||||
case healthy == nil || healthy.Status == metav1.ConditionUnknown:
|
||||
// Provisioned but no health verdict yet: still being brought into
|
||||
// service.
|
||||
return crawlv1alpha1.PhaseProvisioning
|
||||
case healthy.Status == metav1.ConditionTrue:
|
||||
return crawlv1alpha1.PhaseReady
|
||||
default:
|
||||
return crawlv1alpha1.PhaseUnhealthy
|
||||
}
|
||||
}
|
||||
|
||||
// patchStatusIfChanged recomputes the derived status fields and issues one
|
||||
// status patch — or none, when nothing changed. This is the only place the
|
||||
// reconciler writes status.
|
||||
func (r *ProxyReconciler) patchStatusIfChanged(ctx context.Context, base, p *crawlv1alpha1.Proxy) error {
|
||||
p.Status.ObservedGeneration = p.Generation
|
||||
p.Status.Phase = computePhase(p)
|
||||
if equality.Semantic.DeepEqual(base.Status, p.Status) {
|
||||
return nil
|
||||
}
|
||||
return r.Status().Patch(ctx, p, client.MergeFrom(base))
|
||||
}
|
||||
112
internal/controller/status_test.go
Normal file
112
internal/controller/status_test.go
Normal file
@@ -0,0 +1,112 @@
|
||||
package controller
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
func TestComputePhase_truthTable(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cond := func(condType string, status metav1.ConditionStatus, reason string) metav1.Condition {
|
||||
return metav1.Condition{Type: condType, Status: status, Reason: reason}
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
deleting bool
|
||||
conditions []metav1.Condition
|
||||
want crawlv1alpha1.ProxyPhase
|
||||
}{
|
||||
{
|
||||
name: "deletionTimestamp wins over everything",
|
||||
deleting: true,
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated),
|
||||
cond(crawlv1alpha1.ConditionHealthy, metav1.ConditionTrue, "Probing"),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseDeleting,
|
||||
},
|
||||
{
|
||||
name: "no conditions is Pending",
|
||||
want: crawlv1alpha1.PhasePending,
|
||||
},
|
||||
{
|
||||
name: "provisioning in progress",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonProvisioning),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseProvisioning,
|
||||
},
|
||||
{
|
||||
name: "replacing counts as provisioning",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonReplacing),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseProvisioning,
|
||||
},
|
||||
{
|
||||
name: "quota exhaustion is not Failed",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonQuotaExceeded),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseProvisioning,
|
||||
},
|
||||
{
|
||||
name: "permanent error is Failed",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionFalse, ReasonPermanentError),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseFailed,
|
||||
},
|
||||
{
|
||||
name: "provisioned without a health verdict stays Provisioning",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseProvisioning,
|
||||
},
|
||||
{
|
||||
name: "provisioned with Healthy Unknown stays Provisioning",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated),
|
||||
cond(crawlv1alpha1.ConditionHealthy, metav1.ConditionUnknown, "NoProbeYet"),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseProvisioning,
|
||||
},
|
||||
{
|
||||
name: "provisioned and healthy is Ready",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated),
|
||||
cond(crawlv1alpha1.ConditionHealthy, metav1.ConditionTrue, "Probing"),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseReady,
|
||||
},
|
||||
{
|
||||
name: "provisioned but unhealthy is Unhealthy",
|
||||
conditions: []metav1.Condition{
|
||||
cond(crawlv1alpha1.ConditionProvisioned, metav1.ConditionTrue, ReasonCreated),
|
||||
cond(crawlv1alpha1.ConditionHealthy, metav1.ConditionFalse, "ProbeFailed"),
|
||||
},
|
||||
want: crawlv1alpha1.PhaseUnhealthy,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := &crawlv1alpha1.Proxy{}
|
||||
p.Status.Conditions = tc.conditions
|
||||
if tc.deleting {
|
||||
now := metav1.Now()
|
||||
p.DeletionTimestamp = &now
|
||||
}
|
||||
if got := computePhase(p); got != tc.want {
|
||||
t.Errorf("computePhase() = %s, want %s", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -49,6 +49,9 @@ var (
|
||||
)
|
||||
|
||||
func TestControllers(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping envtest suite in -short mode")
|
||||
}
|
||||
RegisterFailHandler(Fail)
|
||||
|
||||
RunSpecs(t, "Controller Suite")
|
||||
|
||||
340
internal/health/engine.go
Normal file
340
internal/health/engine.go
Normal file
@@ -0,0 +1,340 @@
|
||||
package health
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"math/rand/v2"
|
||||
"net"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
apimeta "k8s.io/apimachinery/pkg/api/meta"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/event"
|
||||
logf "sigs.k8s.io/controller-runtime/pkg/log"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
// Snapshot is the engine's current verdict for one proxy, read by the
|
||||
// reconciler when it represents health in the Proxy's status.
|
||||
type Snapshot struct {
|
||||
Healthy bool
|
||||
// Latency is the wall time of the most recent successful probe.
|
||||
Latency time.Duration
|
||||
// LastProbe is when the most recent probe (of either outcome) finished.
|
||||
LastProbe time.Time
|
||||
// LastError is the most recent probe failure; empty after a success.
|
||||
LastError string
|
||||
// ConsecutiveFailures is the current failure streak.
|
||||
ConsecutiveFailures int32
|
||||
}
|
||||
|
||||
// state is the engine's threshold bookkeeping for one proxy. The reported*
|
||||
// fields track what has been delivered over Events; they only advance when a
|
||||
// send succeeds, so a dropped event is retried after the next probe.
|
||||
type state struct {
|
||||
uid types.UID
|
||||
inFlight bool
|
||||
nextDue time.Time
|
||||
|
||||
healthy *bool // nil until a first verdict exists
|
||||
consecOK int32
|
||||
consecFail int32
|
||||
latency time.Duration
|
||||
lastProbe time.Time
|
||||
lastErr string
|
||||
|
||||
reportedHealthy *bool
|
||||
reportedLatency time.Duration
|
||||
lastReport time.Time
|
||||
}
|
||||
|
||||
type probeJob struct {
|
||||
key types.NamespacedName
|
||||
uid types.UID
|
||||
proxyURL *url.URL
|
||||
hc crawlv1alpha1.HealthCheckSpec
|
||||
interval time.Duration
|
||||
}
|
||||
|
||||
// Engine runs the probe scheduler and worker pool as a manager Runnable. It
|
||||
// never writes Proxy status itself — keeping the reconciler the single
|
||||
// status writer — and instead emits a GenericEvent per status-affecting
|
||||
// transition, which the reconciler consumes via source.Channel.
|
||||
type Engine struct {
|
||||
// Reader lists Proxies from the manager's cache each tick.
|
||||
Reader client.Reader
|
||||
// Events carries one enqueue-request per status-affecting transition.
|
||||
// Sends are non-blocking: a wedged reconciler must never stall probing.
|
||||
Events chan event.GenericEvent
|
||||
|
||||
// Workers is the probe worker pool size (default 8).
|
||||
Workers int
|
||||
// Tick is the scheduler interval (default 1s). At tens of proxies a
|
||||
// per-second list scan is free; a timer wheel would be unjustified.
|
||||
Tick time.Duration
|
||||
// MinReportInterval rate-limits latency-only status reports (default 60s).
|
||||
MinReportInterval time.Duration
|
||||
// LatencyFloor is the absolute change below which a latency move is
|
||||
// never status-affecting (default 20ms), so a proxy jittering around a
|
||||
// small latency doesn't write status forever.
|
||||
LatencyFloor time.Duration
|
||||
// ProbeTLSConfig overrides TLS verification for https probe URLs; nil
|
||||
// means system roots. Needed for private CAs (and tests).
|
||||
ProbeTLSConfig *tls.Config
|
||||
|
||||
probeFn func(context.Context, *url.URL, crawlv1alpha1.HealthCheckSpec, *tls.Config) probeResult
|
||||
|
||||
mu sync.Mutex
|
||||
states map[types.NamespacedName]*state
|
||||
}
|
||||
|
||||
// NewEngine returns an Engine with a buffered Events channel, ready to be
|
||||
// handed to both mgr.Add and the reconciler (Health + HealthEvents fields).
|
||||
func NewEngine(reader client.Reader) *Engine {
|
||||
e := &Engine{Reader: reader}
|
||||
e.applyDefaults()
|
||||
return e
|
||||
}
|
||||
|
||||
func (e *Engine) applyDefaults() {
|
||||
if e.Events == nil {
|
||||
e.Events = make(chan event.GenericEvent, 64)
|
||||
}
|
||||
if e.Workers == 0 {
|
||||
e.Workers = 8
|
||||
}
|
||||
if e.Tick == 0 {
|
||||
e.Tick = time.Second
|
||||
}
|
||||
if e.MinReportInterval == 0 {
|
||||
e.MinReportInterval = time.Minute
|
||||
}
|
||||
if e.LatencyFloor == 0 {
|
||||
e.LatencyFloor = 20 * time.Millisecond
|
||||
}
|
||||
if e.probeFn == nil {
|
||||
e.probeFn = probe
|
||||
}
|
||||
if e.states == nil {
|
||||
e.states = map[types.NamespacedName]*state{}
|
||||
}
|
||||
}
|
||||
|
||||
// NeedLeaderElection makes the engine run only on the leader: probing from
|
||||
// every replica would multiply load on the proxies, and only the leader's
|
||||
// reconciler can represent the results anyway.
|
||||
func (e *Engine) NeedLeaderElection() bool { return true }
|
||||
|
||||
// Start runs the scheduler tick loop and the worker pool until ctx ends.
|
||||
func (e *Engine) Start(ctx context.Context) error {
|
||||
e.applyDefaults()
|
||||
jobs := make(chan probeJob)
|
||||
var wg sync.WaitGroup
|
||||
for range e.Workers {
|
||||
wg.Go(func() {
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case job := <-jobs:
|
||||
res := e.probeFn(ctx, job.proxyURL, job.hc, e.ProbeTLSConfig)
|
||||
e.record(job, res, time.Now())
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
ticker := time.NewTicker(e.Tick)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
wg.Wait()
|
||||
return nil
|
||||
case now := <-ticker.C:
|
||||
e.tick(ctx, now, jobs)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// tick lists proxies from the cache, refreshes the state map (create, seed,
|
||||
// prune, UID-mismatch reset), and hands due proxies to the worker pool.
|
||||
func (e *Engine) tick(ctx context.Context, now time.Time, jobs chan<- probeJob) {
|
||||
var list crawlv1alpha1.ProxyList
|
||||
if err := e.Reader.List(ctx, &list); err != nil {
|
||||
logf.FromContext(ctx).Error(err, "health engine: listing proxies")
|
||||
return
|
||||
}
|
||||
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
|
||||
probeable := make(map[types.NamespacedName]struct{}, len(list.Items))
|
||||
for i := range list.Items {
|
||||
p := &list.Items[i]
|
||||
host := p.EffectiveHost()
|
||||
if host == "" || !p.DeletionTimestamp.IsZero() {
|
||||
// Not probeable (provisioning, being replaced, or deleting).
|
||||
// Its state gets pruned below, so a replacement instance starts
|
||||
// with fresh counters.
|
||||
continue
|
||||
}
|
||||
key := client.ObjectKeyFromObject(p)
|
||||
probeable[key] = struct{}{}
|
||||
|
||||
hc := p.HealthCheckOrDefault()
|
||||
interval := time.Duration(hc.IntervalSeconds) * time.Second
|
||||
|
||||
st := e.states[key]
|
||||
if st == nil || st.uid != p.UID {
|
||||
// New proxy, or a delete+recreate under the same name — never
|
||||
// inherit the old object's counters.
|
||||
st = newState(p, now, interval)
|
||||
e.states[key] = st
|
||||
}
|
||||
if st.inFlight || now.Before(st.nextDue) {
|
||||
continue
|
||||
}
|
||||
job := probeJob{
|
||||
key: key,
|
||||
uid: p.UID,
|
||||
proxyURL: &url.URL{Scheme: "http", Host: net.JoinHostPort(host, strconv.Itoa(int(p.EffectivePort())))},
|
||||
hc: hc,
|
||||
interval: interval,
|
||||
}
|
||||
select {
|
||||
case jobs <- job:
|
||||
st.inFlight = true
|
||||
default:
|
||||
// Worker pool saturated; the proxy stays due and is retried on
|
||||
// the next tick.
|
||||
}
|
||||
}
|
||||
|
||||
for key := range e.states {
|
||||
if _, ok := probeable[key]; !ok {
|
||||
delete(e.states, key)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// newState seeds bookkeeping for a proxy the engine hasn't tracked yet. If
|
||||
// the CR already carries a Healthy verdict (leader handover, operator
|
||||
// restart), the verdict is kept — so a healthy proxy doesn't flap to
|
||||
// unknown — counters stay at zero so a real transition still needs a full
|
||||
// threshold run, and the first probe is jittered across the interval so a
|
||||
// restart doesn't fire the whole fleet's probes at once. A proxy with no
|
||||
// prior verdict is probed immediately.
|
||||
func newState(p *crawlv1alpha1.Proxy, now time.Time, interval time.Duration) *state {
|
||||
st := &state{uid: p.UID, nextDue: now}
|
||||
cond := apimeta.FindStatusCondition(p.Status.Conditions, crawlv1alpha1.ConditionHealthy)
|
||||
if cond == nil || cond.Status == metav1.ConditionUnknown {
|
||||
return st
|
||||
}
|
||||
healthy := cond.Status == metav1.ConditionTrue
|
||||
reported := healthy
|
||||
st.healthy = &healthy
|
||||
st.reportedHealthy = &reported
|
||||
st.latency = time.Duration(p.Status.LatencyMillis) * time.Millisecond
|
||||
st.reportedLatency = st.latency
|
||||
st.nextDue = now.Add(rand.N(interval))
|
||||
return st
|
||||
}
|
||||
|
||||
// record folds one probe result into the proxy's threshold state and emits
|
||||
// an event when the result is status-affecting: a first-ever verdict, a
|
||||
// threshold-crossing flip, or a material latency change (beyond
|
||||
// max(LatencyFloor, 50% of reported) and rate-limited by MinReportInterval).
|
||||
func (e *Engine) record(job probeJob, res probeResult, now time.Time) {
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
|
||||
st := e.states[job.key]
|
||||
if st == nil || st.uid != job.uid {
|
||||
return // pruned or replaced while the probe was in flight
|
||||
}
|
||||
st.inFlight = false
|
||||
st.nextDue = now.Add(job.interval)
|
||||
st.lastProbe = now
|
||||
|
||||
if res.ok {
|
||||
st.consecOK++
|
||||
st.consecFail = 0
|
||||
st.latency = res.latency
|
||||
st.lastErr = ""
|
||||
} else {
|
||||
st.consecFail++
|
||||
st.consecOK = 0
|
||||
st.lastErr = res.err.Error()
|
||||
}
|
||||
|
||||
switch {
|
||||
case st.healthy == nil:
|
||||
healthy := res.ok
|
||||
st.healthy = &healthy
|
||||
case *st.healthy && st.consecFail >= job.hc.FailureThreshold:
|
||||
healthy := false
|
||||
st.healthy = &healthy
|
||||
case !*st.healthy && st.consecOK >= job.hc.SuccessThreshold:
|
||||
healthy := true
|
||||
st.healthy = &healthy
|
||||
}
|
||||
|
||||
var emit bool
|
||||
switch {
|
||||
case st.reportedHealthy == nil:
|
||||
emit = true
|
||||
case *st.reportedHealthy != *st.healthy:
|
||||
emit = true
|
||||
case res.ok && *st.healthy:
|
||||
// Latency-only updates matter only for a healthy verdict; a success
|
||||
// streak still below successThreshold must stay silent.
|
||||
delta := st.latency - st.reportedLatency
|
||||
if delta < 0 {
|
||||
delta = -delta
|
||||
}
|
||||
emit = delta > max(e.LatencyFloor, st.reportedLatency/2) &&
|
||||
now.Sub(st.lastReport) > e.MinReportInterval
|
||||
}
|
||||
if !emit {
|
||||
return
|
||||
}
|
||||
|
||||
evt := event.GenericEvent{Object: &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Namespace: job.key.Namespace, Name: job.key.Name},
|
||||
}}
|
||||
select {
|
||||
case e.Events <- evt:
|
||||
reported := *st.healthy
|
||||
st.reportedHealthy = &reported
|
||||
st.reportedLatency = st.latency
|
||||
st.lastReport = now
|
||||
default:
|
||||
// Channel full (reconciler wedged): drop, and deliberately do not
|
||||
// advance the reported markers, so the next probe retries the emit.
|
||||
}
|
||||
}
|
||||
|
||||
// Snapshot returns the engine's current verdict for key; ok is false while
|
||||
// no verdict exists (never probed, or state was reset).
|
||||
func (e *Engine) Snapshot(key types.NamespacedName) (Snapshot, bool) {
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
st := e.states[key]
|
||||
if st == nil || st.healthy == nil {
|
||||
return Snapshot{}, false
|
||||
}
|
||||
return Snapshot{
|
||||
Healthy: *st.healthy,
|
||||
Latency: st.latency,
|
||||
LastProbe: st.lastProbe,
|
||||
LastError: st.lastErr,
|
||||
ConsecutiveFailures: st.consecFail,
|
||||
}, true
|
||||
}
|
||||
372
internal/health/engine_test.go
Normal file
372
internal/health/engine_test.go
Normal file
@@ -0,0 +1,372 @@
|
||||
package health
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"errors"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
"sigs.k8s.io/controller-runtime/pkg/event"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
var testKey = types.NamespacedName{Namespace: "default", Name: "p1"}
|
||||
|
||||
func testEngine() *Engine {
|
||||
e := &Engine{Events: make(chan event.GenericEvent, 8)}
|
||||
e.applyDefaults()
|
||||
return e
|
||||
}
|
||||
|
||||
func testJob(failureThreshold, successThreshold int32) probeJob {
|
||||
return probeJob{
|
||||
key: testKey,
|
||||
uid: "uid-1",
|
||||
hc: crawlv1alpha1.HealthCheckSpec{
|
||||
FailureThreshold: failureThreshold,
|
||||
SuccessThreshold: successThreshold,
|
||||
},
|
||||
interval: 30 * time.Second,
|
||||
}
|
||||
}
|
||||
|
||||
func drainOneEvent(t *testing.T, e *Engine) event.GenericEvent {
|
||||
t.Helper()
|
||||
select {
|
||||
case evt := <-e.Events:
|
||||
return evt
|
||||
default:
|
||||
t.Fatal("expected an event, channel is empty")
|
||||
return event.GenericEvent{}
|
||||
}
|
||||
}
|
||||
|
||||
func assertNoEvent(t *testing.T, e *Engine) {
|
||||
t.Helper()
|
||||
select {
|
||||
case <-e.Events:
|
||||
t.Fatal("unexpected event emitted")
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
func boolPtr(b bool) *bool { return &b }
|
||||
|
||||
func TestRecord_firstResultEmits(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.states[testKey] = &state{uid: "uid-1"}
|
||||
|
||||
e.record(testJob(3, 1), probeResult{ok: true, latency: 30 * time.Millisecond}, time.Now())
|
||||
|
||||
evt := drainOneEvent(t, e)
|
||||
if got := evt.Object.GetName(); got != "p1" {
|
||||
t.Errorf("event object name = %q, want p1", got)
|
||||
}
|
||||
snap, ok := e.Snapshot(testKey)
|
||||
if !ok || !snap.Healthy {
|
||||
t.Errorf("Snapshot = %+v, %v; want healthy verdict", snap, ok)
|
||||
}
|
||||
if snap.Latency != 30*time.Millisecond {
|
||||
t.Errorf("latency = %v, want 30ms", snap.Latency)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecord_failureThresholdFlips(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.states[testKey] = &state{uid: "uid-1", healthy: boolPtr(true), reportedHealthy: boolPtr(true)}
|
||||
job := testJob(3, 1)
|
||||
probeErr := probeResult{err: errors.New("connect refused")}
|
||||
|
||||
e.record(job, probeErr, time.Now())
|
||||
e.record(job, probeErr, time.Now())
|
||||
assertNoEvent(t, e)
|
||||
if snap, _ := e.Snapshot(testKey); !snap.Healthy {
|
||||
t.Fatal("flipped unhealthy before failureThreshold was reached")
|
||||
}
|
||||
|
||||
e.record(job, probeErr, time.Now())
|
||||
drainOneEvent(t, e)
|
||||
snap, _ := e.Snapshot(testKey)
|
||||
if snap.Healthy {
|
||||
t.Error("still healthy after failureThreshold consecutive failures")
|
||||
}
|
||||
if snap.ConsecutiveFailures != 3 {
|
||||
t.Errorf("ConsecutiveFailures = %d, want 3", snap.ConsecutiveFailures)
|
||||
}
|
||||
if !strings.Contains(snap.LastError, "connect refused") {
|
||||
t.Errorf("LastError = %q, want the probe error", snap.LastError)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecord_successThresholdFlips(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.states[testKey] = &state{uid: "uid-1", healthy: boolPtr(false), reportedHealthy: boolPtr(false)}
|
||||
job := testJob(3, 2)
|
||||
success := probeResult{ok: true, latency: 25 * time.Millisecond}
|
||||
|
||||
e.record(job, success, time.Now())
|
||||
assertNoEvent(t, e)
|
||||
|
||||
e.record(job, success, time.Now())
|
||||
drainOneEvent(t, e)
|
||||
if snap, _ := e.Snapshot(testKey); !snap.Healthy {
|
||||
t.Error("not healthy after successThreshold consecutive successes")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecord_latencySuppression(t *testing.T) {
|
||||
t.Parallel()
|
||||
now := time.Now()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
reportedLatency time.Duration
|
||||
lastReport time.Time
|
||||
newLatency time.Duration
|
||||
wantEmit bool
|
||||
}{
|
||||
{
|
||||
name: "small change under the relative floor is suppressed",
|
||||
reportedLatency: 100 * time.Millisecond,
|
||||
lastReport: now.Add(-2 * time.Minute),
|
||||
newLatency: 110 * time.Millisecond,
|
||||
wantEmit: false,
|
||||
},
|
||||
{
|
||||
name: "small absolute jitter at low latency is suppressed",
|
||||
reportedLatency: 5 * time.Millisecond,
|
||||
lastReport: now.Add(-2 * time.Minute),
|
||||
newLatency: 20 * time.Millisecond, // >50% but under the 20ms floor
|
||||
wantEmit: false,
|
||||
},
|
||||
{
|
||||
name: "material change after the rate window emits",
|
||||
reportedLatency: 100 * time.Millisecond,
|
||||
lastReport: now.Add(-2 * time.Minute),
|
||||
newLatency: 200 * time.Millisecond,
|
||||
wantEmit: true,
|
||||
},
|
||||
{
|
||||
name: "material change inside the rate window is suppressed",
|
||||
reportedLatency: 100 * time.Millisecond,
|
||||
lastReport: now.Add(-10 * time.Second),
|
||||
newLatency: 400 * time.Millisecond,
|
||||
wantEmit: false,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.states[testKey] = &state{
|
||||
uid: "uid-1",
|
||||
healthy: boolPtr(true),
|
||||
reportedHealthy: boolPtr(true),
|
||||
reportedLatency: tc.reportedLatency,
|
||||
lastReport: tc.lastReport,
|
||||
}
|
||||
e.record(testJob(3, 1), probeResult{ok: true, latency: tc.newLatency}, now)
|
||||
if tc.wantEmit {
|
||||
drainOneEvent(t, e)
|
||||
} else {
|
||||
assertNoEvent(t, e)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecord_droppedEventIsRetried(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.Events = make(chan event.GenericEvent) // unbuffered, nobody reading
|
||||
e.states[testKey] = &state{uid: "uid-1"}
|
||||
success := probeResult{ok: true, latency: 30 * time.Millisecond}
|
||||
|
||||
e.record(testJob(3, 1), success, time.Now())
|
||||
e.mu.Lock()
|
||||
reported := e.states[testKey].reportedHealthy
|
||||
e.mu.Unlock()
|
||||
if reported != nil {
|
||||
t.Fatal("reported marker advanced although the event was dropped")
|
||||
}
|
||||
|
||||
// Channel drains (reconciler recovers): the next probe re-emits.
|
||||
e.Events = make(chan event.GenericEvent, 1)
|
||||
e.record(testJob(3, 1), success, time.Now())
|
||||
drainOneEvent(t, e)
|
||||
}
|
||||
|
||||
func TestRecord_staleJobIsIgnored(t *testing.T) {
|
||||
t.Parallel()
|
||||
e := testEngine()
|
||||
e.states[testKey] = &state{uid: "uid-NEW"}
|
||||
|
||||
job := testJob(3, 1)
|
||||
job.uid = "uid-OLD"
|
||||
e.record(job, probeResult{ok: true, latency: time.Millisecond}, time.Now())
|
||||
|
||||
assertNoEvent(t, e)
|
||||
if _, ok := e.Snapshot(testKey); ok {
|
||||
t.Error("stale probe produced a verdict for the new object")
|
||||
}
|
||||
}
|
||||
|
||||
func externalProxy(name, host string, mut ...func(*crawlv1alpha1.Proxy)) *crawlv1alpha1.Proxy {
|
||||
p := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: name, Namespace: "default", UID: types.UID("uid-" + name),
|
||||
},
|
||||
Spec: crawlv1alpha1.ProxySpec{
|
||||
Mode: crawlv1alpha1.ModeExternal,
|
||||
Endpoint: &crawlv1alpha1.EndpointSpec{Host: host, Port: 3128},
|
||||
},
|
||||
}
|
||||
for _, m := range mut {
|
||||
m(p)
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
func TestTick_schedulingAndPruning(t *testing.T) {
|
||||
t.Parallel()
|
||||
s := runtime.NewScheme()
|
||||
if err := crawlv1alpha1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("scheme: %v", err)
|
||||
}
|
||||
|
||||
probeable := externalProxy("probeable", "10.0.0.1")
|
||||
seeded := externalProxy("seeded", "10.0.0.2", func(p *crawlv1alpha1.Proxy) {
|
||||
p.Status.Conditions = []metav1.Condition{{
|
||||
Type: crawlv1alpha1.ConditionHealthy, Status: metav1.ConditionTrue,
|
||||
Reason: "ProbeSucceeded", LastTransitionTime: metav1.Now(),
|
||||
}}
|
||||
p.Status.LatencyMillis = 42
|
||||
})
|
||||
noIP := &crawlv1alpha1.Proxy{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "no-ip", Namespace: "default", UID: "uid-no-ip"},
|
||||
Spec: crawlv1alpha1.ProxySpec{Mode: crawlv1alpha1.ModeManaged, Provider: "stub"},
|
||||
}
|
||||
|
||||
e := testEngine()
|
||||
e.Reader = fake.NewClientBuilder().WithScheme(s).
|
||||
WithObjects(probeable, seeded, noIP).Build()
|
||||
// Stale entries: one for a proxy that no longer exists, one under a key
|
||||
// that now belongs to a different UID (delete + recreate).
|
||||
e.states[types.NamespacedName{Namespace: "default", Name: "gone"}] = &state{uid: "uid-gone"}
|
||||
e.states[types.NamespacedName{Namespace: "default", Name: "probeable"}] = &state{
|
||||
uid: "uid-previous-incarnation", healthy: boolPtr(false),
|
||||
}
|
||||
|
||||
jobs := make(chan probeJob, 8)
|
||||
e.tick(context.Background(), time.Now(), jobs)
|
||||
|
||||
var dispatched []probeJob
|
||||
for {
|
||||
select {
|
||||
case j := <-jobs:
|
||||
dispatched = append(dispatched, j)
|
||||
continue
|
||||
default:
|
||||
}
|
||||
break
|
||||
}
|
||||
|
||||
if len(dispatched) != 1 {
|
||||
t.Fatalf("dispatched %d jobs, want exactly 1 (only the fresh probeable proxy)", len(dispatched))
|
||||
}
|
||||
j := dispatched[0]
|
||||
if j.key.Name != "probeable" || j.uid != "uid-probeable" {
|
||||
t.Errorf("dispatched job = %+v, want the recreated probeable proxy", j)
|
||||
}
|
||||
if want := "http://" + "10.0.0.1:" + strconv.Itoa(3128); j.proxyURL.String() != want {
|
||||
t.Errorf("proxyURL = %s, want %s", j.proxyURL, want)
|
||||
}
|
||||
|
||||
e.mu.Lock()
|
||||
defer e.mu.Unlock()
|
||||
if _, ok := e.states[types.NamespacedName{Namespace: "default", Name: "gone"}]; ok {
|
||||
t.Error("state for a deleted proxy was not pruned")
|
||||
}
|
||||
if _, ok := e.states[types.NamespacedName{Namespace: "default", Name: "no-ip"}]; ok {
|
||||
t.Error("state was created for a proxy with no IP")
|
||||
}
|
||||
st := e.states[types.NamespacedName{Namespace: "default", Name: "probeable"}]
|
||||
if st == nil || st.uid != "uid-probeable" {
|
||||
t.Fatalf("state for recreated proxy = %+v, want fresh state with the new UID", st)
|
||||
}
|
||||
if st.healthy != nil && !*st.healthy {
|
||||
t.Error("recreated proxy inherited the previous incarnation's unhealthy verdict")
|
||||
}
|
||||
seededSt := e.states[types.NamespacedName{Namespace: "default", Name: "seeded"}]
|
||||
if seededSt == nil {
|
||||
t.Fatal("no state created for the seeded proxy")
|
||||
}
|
||||
if seededSt.healthy == nil || !*seededSt.healthy {
|
||||
t.Error("seeded proxy did not inherit its Healthy condition")
|
||||
}
|
||||
if seededSt.reportedHealthy == nil || !*seededSt.reportedHealthy {
|
||||
t.Error("seeded verdict must count as already reported, or restart would re-emit for the whole fleet")
|
||||
}
|
||||
if seededSt.reportedLatency != 42*time.Millisecond {
|
||||
t.Errorf("seeded reportedLatency = %v, want 42ms", seededSt.reportedLatency)
|
||||
}
|
||||
if seededSt.consecOK != 0 || seededSt.consecFail != 0 {
|
||||
t.Error("seeded counters must start at zero")
|
||||
}
|
||||
}
|
||||
|
||||
func TestEngine_StartEndToEnd(t *testing.T) {
|
||||
t.Parallel()
|
||||
s := runtime.NewScheme()
|
||||
if err := crawlv1alpha1.AddToScheme(s); err != nil {
|
||||
t.Fatalf("scheme: %v", err)
|
||||
}
|
||||
|
||||
e := testEngine()
|
||||
e.Tick = 5 * time.Millisecond
|
||||
e.Reader = fake.NewClientBuilder().WithScheme(s).
|
||||
WithObjects(externalProxy("p1", "192.0.2.1")).Build()
|
||||
e.probeFn = func(context.Context, *url.URL, crawlv1alpha1.HealthCheckSpec, *tls.Config) probeResult {
|
||||
return probeResult{ok: true, latency: 12 * time.Millisecond}
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- e.Start(ctx) }()
|
||||
|
||||
select {
|
||||
case evt := <-e.Events:
|
||||
if evt.Object.GetName() != "p1" {
|
||||
t.Errorf("event for %q, want p1", evt.Object.GetName())
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("no health event within 5s")
|
||||
}
|
||||
snap, ok := e.Snapshot(types.NamespacedName{Namespace: "default", Name: "p1"})
|
||||
if !ok || !snap.Healthy || snap.Latency != 12*time.Millisecond {
|
||||
t.Errorf("Snapshot = %+v, %v; want healthy at 12ms", snap, ok)
|
||||
}
|
||||
|
||||
cancel()
|
||||
select {
|
||||
case err := <-done:
|
||||
if err != nil {
|
||||
t.Errorf("Start returned %v, want nil on context cancel", err)
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("Start did not stop within 5s of cancel")
|
||||
}
|
||||
}
|
||||
66
internal/health/probe.go
Normal file
66
internal/health/probe.go
Normal file
@@ -0,0 +1,66 @@
|
||||
// Package health actively probes every proxy by fetching a URL through the
|
||||
// proxy itself, keeps per-proxy threshold state, and pushes status-affecting
|
||||
// transitions to the reconciler over a channel. The engine owns health
|
||||
// state; the reconciler owns its representation in the Proxy's status.
|
||||
package health
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"fmt"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"slices"
|
||||
"time"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
// probeResult is the outcome of a single through-the-proxy probe.
|
||||
type probeResult struct {
|
||||
ok bool
|
||||
latency time.Duration
|
||||
err error
|
||||
}
|
||||
|
||||
// probe fetches hc.ProbeURL through the proxy at proxyURL. For an https
|
||||
// probe URL the transport issues CONNECT to the proxy and TLS-handshakes
|
||||
// through the tunnel; a proxy that accepts TCP but cannot egress answers
|
||||
// CONNECT with a non-200, which client.Do surfaces as an error, not a
|
||||
// response — so success requires err == nil AND an expected status code.
|
||||
// tlsCfg is nil in production (system roots); tests and private-CA setups
|
||||
// inject their own.
|
||||
func probe(ctx context.Context, proxyURL *url.URL, hc crawlv1alpha1.HealthCheckSpec, tlsCfg *tls.Config) probeResult {
|
||||
timeout := time.Duration(hc.TimeoutSeconds) * time.Second
|
||||
transport := &http.Transport{
|
||||
Proxy: http.ProxyURL(proxyURL),
|
||||
// Load-bearing: with keep-alives on, net/http caches the established
|
||||
// CONNECT tunnel and later probes would never re-exercise CONNECT —
|
||||
// exactly the failure this probe exists to catch.
|
||||
DisableKeepAlives: true,
|
||||
ForceAttemptHTTP2: false,
|
||||
TLSHandshakeTimeout: timeout,
|
||||
ResponseHeaderTimeout: timeout,
|
||||
TLSClientConfig: tlsCfg,
|
||||
DialContext: (&net.Dialer{Timeout: timeout}).DialContext,
|
||||
}
|
||||
defer transport.CloseIdleConnections()
|
||||
|
||||
client := &http.Client{Transport: transport, Timeout: timeout}
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, hc.ProbeURL, nil)
|
||||
if err != nil {
|
||||
return probeResult{err: fmt.Errorf("building probe request: %w", err)}
|
||||
}
|
||||
start := time.Now()
|
||||
resp, err := client.Do(req)
|
||||
latency := time.Since(start)
|
||||
if err != nil {
|
||||
return probeResult{latency: latency, err: err}
|
||||
}
|
||||
defer func() { _ = resp.Body.Close() }()
|
||||
if !slices.Contains(hc.ExpectedStatusCodes, int32(resp.StatusCode)) {
|
||||
return probeResult{latency: latency, err: fmt.Errorf("unexpected status %d", resp.StatusCode)}
|
||||
}
|
||||
return probeResult{ok: true, latency: latency}
|
||||
}
|
||||
169
internal/health/probe_test.go
Normal file
169
internal/health/probe_test.go
Normal file
@@ -0,0 +1,169 @@
|
||||
package health
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"crypto/x509"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/url"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
crawlv1alpha1 "gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/api/v1alpha1"
|
||||
)
|
||||
|
||||
// startConnectProxy runs a minimal but real HTTP proxy: CONNECT tunneling
|
||||
// for https targets, absolute-URI forwarding for plain http ones. With
|
||||
// refuseConnect it answers CONNECT with 502 — the "accepts TCP but cannot
|
||||
// egress" failure mode the probe must classify as unhealthy.
|
||||
func startConnectProxy(t *testing.T, refuseConnect bool) *httptest.Server {
|
||||
t.Helper()
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method == http.MethodConnect {
|
||||
if refuseConnect {
|
||||
http.Error(w, "no egress", http.StatusBadGateway)
|
||||
return
|
||||
}
|
||||
dst, err := net.DialTimeout("tcp", r.Host, time.Second)
|
||||
if err != nil {
|
||||
http.Error(w, err.Error(), http.StatusBadGateway)
|
||||
return
|
||||
}
|
||||
conn, bufrw, err := http.NewResponseController(w).Hijack()
|
||||
if err != nil {
|
||||
_ = dst.Close()
|
||||
t.Errorf("hijack: %v", err)
|
||||
return
|
||||
}
|
||||
_, _ = bufrw.WriteString("HTTP/1.1 200 Connection established\r\n\r\n")
|
||||
_ = bufrw.Flush()
|
||||
done := make(chan struct{}, 2)
|
||||
go func() { _, _ = io.Copy(dst, bufrw); done <- struct{}{} }()
|
||||
go func() { _, _ = io.Copy(conn, dst); done <- struct{}{} }()
|
||||
<-done
|
||||
_ = conn.Close()
|
||||
_ = dst.Close()
|
||||
return
|
||||
}
|
||||
out := r.Clone(r.Context())
|
||||
out.RequestURI = ""
|
||||
resp, err := http.DefaultTransport.RoundTrip(out)
|
||||
if err != nil {
|
||||
http.Error(w, err.Error(), http.StatusBadGateway)
|
||||
return
|
||||
}
|
||||
defer func() { _ = resp.Body.Close() }()
|
||||
for k, vv := range resp.Header {
|
||||
for _, v := range vv {
|
||||
w.Header().Add(k, v)
|
||||
}
|
||||
}
|
||||
w.WriteHeader(resp.StatusCode)
|
||||
_, _ = io.Copy(w, resp.Body)
|
||||
}))
|
||||
t.Cleanup(srv.Close)
|
||||
return srv
|
||||
}
|
||||
|
||||
func startTLSTarget(t *testing.T, status int) (*httptest.Server, *tls.Config) {
|
||||
t.Helper()
|
||||
target := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.WriteHeader(status)
|
||||
}))
|
||||
t.Cleanup(target.Close)
|
||||
pool := x509.NewCertPool()
|
||||
pool.AddCert(target.Certificate())
|
||||
return target, &tls.Config{RootCAs: pool}
|
||||
}
|
||||
|
||||
func proxyURL(t *testing.T, srv *httptest.Server) *url.URL {
|
||||
t.Helper()
|
||||
u, err := url.Parse(srv.URL)
|
||||
if err != nil {
|
||||
t.Fatalf("parsing proxy URL: %v", err)
|
||||
}
|
||||
return u
|
||||
}
|
||||
|
||||
func testHC(probeTarget string) crawlv1alpha1.HealthCheckSpec {
|
||||
return crawlv1alpha1.HealthCheckSpec{
|
||||
ProbeURL: probeTarget,
|
||||
IntervalSeconds: 30,
|
||||
TimeoutSeconds: 5,
|
||||
FailureThreshold: 3,
|
||||
SuccessThreshold: 1,
|
||||
ExpectedStatusCodes: []int32{200, 204},
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbe_connectTunnelSucceeds(t *testing.T) {
|
||||
t.Parallel()
|
||||
target, tlsCfg := startTLSTarget(t, http.StatusNoContent)
|
||||
proxy := startConnectProxy(t, false)
|
||||
|
||||
res := probe(context.Background(), proxyURL(t, proxy), testHC(target.URL), tlsCfg)
|
||||
if !res.ok {
|
||||
t.Fatalf("probe failed through working proxy: %v", res.err)
|
||||
}
|
||||
if res.latency <= 0 {
|
||||
t.Errorf("latency = %v, want > 0", res.latency)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbe_refusedConnectFails(t *testing.T) {
|
||||
t.Parallel()
|
||||
target, tlsCfg := startTLSTarget(t, http.StatusNoContent)
|
||||
proxy := startConnectProxy(t, true)
|
||||
|
||||
res := probe(context.Background(), proxyURL(t, proxy), testHC(target.URL), tlsCfg)
|
||||
if res.ok {
|
||||
t.Fatal("probe succeeded through a proxy that refuses CONNECT")
|
||||
}
|
||||
if res.err == nil {
|
||||
t.Error("expected an error from the refused CONNECT")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbe_unexpectedStatusFails(t *testing.T) {
|
||||
t.Parallel()
|
||||
target, tlsCfg := startTLSTarget(t, http.StatusInternalServerError)
|
||||
proxy := startConnectProxy(t, false)
|
||||
|
||||
res := probe(context.Background(), proxyURL(t, proxy), testHC(target.URL), tlsCfg)
|
||||
if res.ok {
|
||||
t.Fatal("probe succeeded on a 500 response")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbe_unreachableProxyFails(t *testing.T) {
|
||||
t.Parallel()
|
||||
// A listener that is immediately closed: guaranteed-refused port.
|
||||
l, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("reserving port: %v", err)
|
||||
}
|
||||
dead := &url.URL{Scheme: "http", Host: l.Addr().String()}
|
||||
_ = l.Close()
|
||||
|
||||
res := probe(context.Background(), dead, testHC("https://example.invalid/"), nil)
|
||||
if res.ok {
|
||||
t.Fatal("probe succeeded against a dead proxy")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbe_plainHTTPForwardSucceeds(t *testing.T) {
|
||||
t.Parallel()
|
||||
target := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}))
|
||||
t.Cleanup(target.Close)
|
||||
proxy := startConnectProxy(t, false)
|
||||
|
||||
res := probe(context.Background(), proxyURL(t, proxy), testHC(target.URL), nil)
|
||||
if !res.ok {
|
||||
t.Fatalf("plain-http probe failed: %v", res.err)
|
||||
}
|
||||
}
|
||||
@@ -7,23 +7,6 @@ import (
|
||||
"sigs.k8s.io/yaml"
|
||||
)
|
||||
|
||||
// Fail-with classes accepted by MockConfig.FailWith, shared with the mock
|
||||
// provider's fault injection so the two sides never drift on the string
|
||||
// values.
|
||||
const (
|
||||
FailWithNotFound = "notfound"
|
||||
FailWithQuota = "quota"
|
||||
FailWithTransient = "transient"
|
||||
FailWithPermanent = "permanent"
|
||||
)
|
||||
|
||||
var validFailClasses = map[string]bool{
|
||||
FailWithNotFound: true,
|
||||
FailWithQuota: true,
|
||||
FailWithTransient: true,
|
||||
FailWithPermanent: true,
|
||||
}
|
||||
|
||||
// Config is the top-level shape of the --providers-config file.
|
||||
type Config struct {
|
||||
Providers []ProviderConfig `json:"providers"`
|
||||
@@ -36,25 +19,15 @@ type Config struct {
|
||||
type ProviderConfig struct {
|
||||
Name string `json:"name"`
|
||||
Type string `json:"type"`
|
||||
Mock *MockConfig `json:"mock,omitempty"`
|
||||
Kubernetes *KubernetesConfig `json:"kubernetes,omitempty"`
|
||||
GCP *GCPConfig `json:"gcp,omitempty"`
|
||||
}
|
||||
|
||||
// MockConfig configures the in-memory mock provider.
|
||||
type MockConfig struct {
|
||||
// ProvisionDelaySeconds is how long a created instance reports
|
||||
// Provisioning before Running. Default 5.
|
||||
ProvisionDelaySeconds int32 `json:"provisionDelaySeconds,omitempty"`
|
||||
// DeleteDelaySeconds is how long a deleted instance reports Terminated
|
||||
// before Get starts returning ErrNotFound. Default 1.
|
||||
DeleteDelaySeconds int32 `json:"deleteDelaySeconds,omitempty"`
|
||||
// FailNextCreates makes the next N Create calls fail with FailWith,
|
||||
// for exercising the reconciler's error handling in demos.
|
||||
FailNextCreates int `json:"failNextCreates,omitempty"`
|
||||
// FailWith selects the error class injected failures return: one of
|
||||
// FailWithNotFound/FailWithQuota/FailWithTransient/FailWithPermanent.
|
||||
// Default FailWithTransient.
|
||||
FailWith string `json:"failWith,omitempty"`
|
||||
// KubernetesConfig configures the kubernetes-pod provider, which creates
|
||||
// proxy Pods in the same cluster the operator itself runs in.
|
||||
type KubernetesConfig struct {
|
||||
// Image is the proxy container image. Default "ubuntu/squid:6.6-24.04_edge".
|
||||
Image string `json:"image,omitempty"`
|
||||
}
|
||||
|
||||
// GCPConfig configures a named GCP provider instance.
|
||||
@@ -109,16 +82,13 @@ func (c *Config) validate() error {
|
||||
seen[p.Name] = true
|
||||
|
||||
switch p.Type {
|
||||
case "mock":
|
||||
case "kubernetes":
|
||||
if p.GCP != nil {
|
||||
return fmt.Errorf("providers[%d] %q: type is mock but a gcp block is set", i, p.Name)
|
||||
}
|
||||
if p.Mock != nil && p.Mock.FailWith != "" && !validFailClasses[p.Mock.FailWith] {
|
||||
return fmt.Errorf("providers[%d] %q: unknown mock.failWith %q", i, p.Name, p.Mock.FailWith)
|
||||
return fmt.Errorf("providers[%d] %q: type is kubernetes but a gcp block is set", i, p.Name)
|
||||
}
|
||||
case "gcp":
|
||||
if p.Mock != nil {
|
||||
return fmt.Errorf("providers[%d] %q: type is gcp but a mock block is set", i, p.Name)
|
||||
if p.Kubernetes != nil {
|
||||
return fmt.Errorf("providers[%d] %q: type is gcp but a kubernetes block is set", i, p.Name)
|
||||
}
|
||||
if p.GCP == nil || p.GCP.Project == "" {
|
||||
return fmt.Errorf("providers[%d] %q: gcp.project is required", i, p.Name)
|
||||
|
||||
@@ -9,10 +9,10 @@ func TestLoadConfig_valid(t *testing.T) {
|
||||
t.Parallel()
|
||||
data := []byte(`
|
||||
providers:
|
||||
- name: mock
|
||||
type: mock
|
||||
mock:
|
||||
provisionDelaySeconds: 2
|
||||
- name: kubernetes
|
||||
type: kubernetes
|
||||
kubernetes:
|
||||
image: ubuntu/squid:6.6-24.04_edge
|
||||
- name: gcp-eu
|
||||
type: gcp
|
||||
gcp:
|
||||
@@ -26,8 +26,8 @@ providers:
|
||||
if len(cfg.Providers) != 2 {
|
||||
t.Fatalf("len(cfg.Providers) = %d, want 2", len(cfg.Providers))
|
||||
}
|
||||
if cfg.Providers[0].Mock == nil || cfg.Providers[0].Mock.ProvisionDelaySeconds != 2 {
|
||||
t.Errorf("providers[0].mock = %+v, want ProvisionDelaySeconds=2", cfg.Providers[0].Mock)
|
||||
if cfg.Providers[0].Kubernetes == nil || cfg.Providers[0].Kubernetes.Image != "ubuntu/squid:6.6-24.04_edge" {
|
||||
t.Errorf("providers[0].kubernetes = %+v, want Image=ubuntu/squid:6.6-24.04_edge", cfg.Providers[0].Kubernetes)
|
||||
}
|
||||
if cfg.Providers[1].GCP == nil || cfg.Providers[1].GCP.Project != "my-project" {
|
||||
t.Errorf("providers[1].gcp = %+v, want Project=my-project", cfg.Providers[1].GCP)
|
||||
@@ -50,17 +50,17 @@ func TestLoadConfig_invalid(t *testing.T) {
|
||||
name: "missing name",
|
||||
yaml: `
|
||||
providers:
|
||||
- type: mock`,
|
||||
- type: kubernetes`,
|
||||
wantErrSub: "name is required",
|
||||
},
|
||||
{
|
||||
name: "duplicate name",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: mock
|
||||
type: mock
|
||||
- name: mock
|
||||
type: mock`,
|
||||
- name: kubernetes
|
||||
type: kubernetes
|
||||
- name: kubernetes
|
||||
type: kubernetes`,
|
||||
wantErrSub: "duplicate provider name",
|
||||
},
|
||||
{
|
||||
@@ -97,43 +97,33 @@ providers:
|
||||
wantErrSub: "gcp.project is required",
|
||||
},
|
||||
{
|
||||
name: "mock type with gcp block",
|
||||
name: "kubernetes type with gcp block",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1
|
||||
type: mock
|
||||
type: kubernetes
|
||||
gcp:
|
||||
project: my-project`,
|
||||
wantErrSub: "type is mock but a gcp block is set",
|
||||
wantErrSub: "type is kubernetes but a gcp block is set",
|
||||
},
|
||||
{
|
||||
name: "gcp type with mock block",
|
||||
name: "gcp type with kubernetes block",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1
|
||||
type: gcp
|
||||
gcp:
|
||||
project: my-project
|
||||
mock:
|
||||
failNextCreates: 1`,
|
||||
wantErrSub: "type is gcp but a mock block is set",
|
||||
},
|
||||
{
|
||||
name: "unknown mock.failWith",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1
|
||||
type: mock
|
||||
mock:
|
||||
failWith: oops`,
|
||||
wantErrSub: `unknown mock.failWith "oops"`,
|
||||
kubernetes:
|
||||
image: custom-image`,
|
||||
wantErrSub: "type is gcp but a kubernetes block is set",
|
||||
},
|
||||
{
|
||||
name: "strict mode rejects unknown top-level key",
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1
|
||||
type: mock
|
||||
type: kubernetes
|
||||
extraneous: true`,
|
||||
wantErrSub: "parsing providers config",
|
||||
},
|
||||
@@ -142,7 +132,7 @@ extraneous: true`,
|
||||
yaml: `
|
||||
providers:
|
||||
- name: p1
|
||||
type: mock
|
||||
type: kubernetes
|
||||
bogus: true`,
|
||||
wantErrSub: "parsing providers config",
|
||||
},
|
||||
|
||||
202
internal/provider/kubernetes/kubernetes.go
Normal file
202
internal/provider/kubernetes/kubernetes.go
Normal file
@@ -0,0 +1,202 @@
|
||||
// Package kubernetes is a provider.Provider that creates real Pods running
|
||||
// a Squid container in the same cluster the operator itself runs in —
|
||||
// unlike a cloud provider, "creating compute" here means talking back to
|
||||
// the very Kubernetes API the operator is already watching.
|
||||
package kubernetes
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"k8s.io/client-go/tools/cache"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
// defaultImage is Canonical's actively maintained Squid image for Ubuntu
|
||||
// 24.04 LTS, verified before picking it: a public image (no build/load
|
||||
// step needed for a kind demo), 50M+ pulls, updated the same day this
|
||||
// decision was made.
|
||||
const defaultImage = "ubuntu/squid:6.6-24.04_edge"
|
||||
|
||||
// Provider creates proxy Pods. It builds its own client rather than
|
||||
// depending on the manager's, via ctrl.GetConfig() — which auto-detects
|
||||
// in-cluster config when running as a Pod and falls back to the local
|
||||
// kubeconfig otherwise. That's what makes `make run` against a local kind
|
||||
// cluster and running in-cluster use the exact same code path with no
|
||||
// provider-specific wiring in cmd/main.go.
|
||||
type Provider struct {
|
||||
client client.Client
|
||||
image string
|
||||
}
|
||||
|
||||
// New builds a kubernetes Provider from its config block. Satisfies
|
||||
// registry.Constructor.
|
||||
func New(_ context.Context, cfg provider.ProviderConfig) (provider.Provider, error) {
|
||||
restCfg, err := ctrl.GetConfig()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("kubernetes provider %q: loading kubeconfig: %w", cfg.Name, err)
|
||||
}
|
||||
scheme := runtime.NewScheme()
|
||||
if err := corev1.AddToScheme(scheme); err != nil {
|
||||
return nil, fmt.Errorf("kubernetes provider %q: %w", cfg.Name, err)
|
||||
}
|
||||
c, err := client.New(restCfg, client.Options{Scheme: scheme})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("kubernetes provider %q: building client: %w", cfg.Name, err)
|
||||
}
|
||||
return newWithClient(c, cfg), nil
|
||||
}
|
||||
|
||||
// newWithClient builds a Provider around an already-constructed client,
|
||||
// bypassing ctrl.GetConfig(). Tests use this exclusively — New() must
|
||||
// never run under `go test`, since ctrl.GetConfig() would happily connect
|
||||
// to whatever real cluster the developer's kubeconfig points at.
|
||||
func newWithClient(c client.Client, cfg provider.ProviderConfig) *Provider {
|
||||
image := defaultImage
|
||||
if cfg.Kubernetes != nil && cfg.Kubernetes.Image != "" {
|
||||
image = cfg.Kubernetes.Image
|
||||
}
|
||||
return &Provider{client: c, image: image}
|
||||
}
|
||||
|
||||
// Create creates a Pod running the proxy container. Idempotent by
|
||||
// req.Name: if a Pod with that name already exists in req.Namespace, its
|
||||
// providerID is returned rather than erroring, so a repeat call after a
|
||||
// crash finds the existing Pod instead of creating a duplicate.
|
||||
func (p *Provider) Create(ctx context.Context, req provider.CreateRequest) (string, error) {
|
||||
pod := buildPod(p.image, req)
|
||||
if err := p.client.Create(ctx, pod); err != nil {
|
||||
if apierrors.IsAlreadyExists(err) {
|
||||
return providerID(req.Namespace, req.Name), nil
|
||||
}
|
||||
return "", classify("create", req.Name, err)
|
||||
}
|
||||
return providerID(req.Namespace, req.Name), nil
|
||||
}
|
||||
|
||||
// Get returns the current state of a previously created Pod.
|
||||
func (p *Provider) Get(ctx context.Context, id string) (*provider.Instance, error) {
|
||||
ns, name, err := parseProviderID(id)
|
||||
if err != nil {
|
||||
return nil, provider.Wrap(provider.ErrPermanent, "get", "kubernetes", id, err)
|
||||
}
|
||||
var pod corev1.Pod
|
||||
if err := p.client.Get(ctx, types.NamespacedName{Namespace: ns, Name: name}, &pod); err != nil {
|
||||
if apierrors.IsNotFound(err) {
|
||||
return nil, provider.Wrap(provider.ErrNotFound, "get", "kubernetes", id, nil)
|
||||
}
|
||||
return nil, classify("get", id, err)
|
||||
}
|
||||
return instanceFromPod(&pod), nil
|
||||
}
|
||||
|
||||
// Delete is idempotent: deleting an unknown Pod is not an error.
|
||||
func (p *Provider) Delete(ctx context.Context, id string) error {
|
||||
ns, name, err := parseProviderID(id)
|
||||
if err != nil {
|
||||
return provider.Wrap(provider.ErrPermanent, "delete", "kubernetes", id, err)
|
||||
}
|
||||
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name}}
|
||||
if err := p.client.Delete(ctx, pod); err != nil && !apierrors.IsNotFound(err) {
|
||||
return classify("delete", id, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ListByTag returns every Pod carrying LabelManaged, across every
|
||||
// namespace — orphan GC needs to find proxy Pods regardless of which
|
||||
// namespace Proxy CRs happen to live in, which is why this provider's RBAC
|
||||
// is cluster-scoped rather than namespaced.
|
||||
func (p *Provider) ListByTag(ctx context.Context) ([]provider.Instance, error) {
|
||||
var pods corev1.PodList
|
||||
if err := p.client.List(ctx, &pods, client.MatchingLabels{
|
||||
provider.LabelManaged: provider.LabelManagedYes,
|
||||
}); err != nil {
|
||||
return nil, classify("list", "", err)
|
||||
}
|
||||
out := make([]provider.Instance, 0, len(pods.Items))
|
||||
for i := range pods.Items {
|
||||
out = append(out, *instanceFromPod(&pods.Items[i]))
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// providerID encodes namespace and name into the opaque providerID string
|
||||
// the Provider interface exposes, so Get/Delete are self-contained without
|
||||
// needing to separately track or re-derive which namespace a Pod lives in
|
||||
// — the same reasoning as the GCP provider's zone-qualified providerID.
|
||||
func providerID(namespace, name string) string {
|
||||
return namespace + "/" + name
|
||||
}
|
||||
|
||||
func parseProviderID(id string) (namespace, name string, err error) {
|
||||
ns, n, err := cache.SplitMetaNamespaceKey(id)
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
if ns == "" {
|
||||
return "", "", fmt.Errorf("providerID %q missing a namespace", id)
|
||||
}
|
||||
return ns, n, nil
|
||||
}
|
||||
|
||||
func instanceFromPod(pod *corev1.Pod) *provider.Instance {
|
||||
inst := &provider.Instance{
|
||||
ID: providerID(pod.Namespace, pod.Name),
|
||||
UID: pod.Labels[provider.LabelUID],
|
||||
CreatedAt: pod.CreationTimestamp.Time,
|
||||
State: stateFromPod(pod),
|
||||
}
|
||||
if inst.State == provider.StateRunning {
|
||||
inst.IP = pod.Status.PodIP
|
||||
}
|
||||
return inst
|
||||
}
|
||||
|
||||
// stateFromPod maps a Pod's phase to InstanceState. Succeeded/Failed/
|
||||
// Unknown all collapse to Terminated: the reconciler treats Stopped and
|
||||
// Terminated identically (delete and recreate — cattle, not pets), so a
|
||||
// finer-grained distinction between "the container exited" and "the node
|
||||
// went unreachable" wouldn't change any behavior.
|
||||
func stateFromPod(pod *corev1.Pod) provider.InstanceState {
|
||||
switch pod.Status.Phase {
|
||||
case corev1.PodPending:
|
||||
return provider.StateProvisioning
|
||||
case corev1.PodRunning:
|
||||
if pod.Status.PodIP == "" {
|
||||
// Never publish an empty IP while the kubelet is still
|
||||
// finishing setup.
|
||||
return provider.StateProvisioning
|
||||
}
|
||||
return provider.StateRunning
|
||||
default: // Succeeded, Failed, Unknown
|
||||
return provider.StateTerminated
|
||||
}
|
||||
}
|
||||
|
||||
// classify maps a Kubernetes API error to the provider error taxonomy.
|
||||
// Quota-exceeded and RBAC-denied both surface as 403 Forbidden from the
|
||||
// API server with nothing in apierrors to tell them apart programmatically
|
||||
// — a known simplification for this prototype; both classify as
|
||||
// ErrPermanent, which is the safer default of the two (stop retrying
|
||||
// rather than hammering an API server that will never allow the request).
|
||||
func classify(op, id string, err error) error {
|
||||
switch {
|
||||
case apierrors.IsNotFound(err):
|
||||
return provider.Wrap(provider.ErrNotFound, op, "kubernetes", id, err)
|
||||
case apierrors.IsTooManyRequests(err), apierrors.IsServerTimeout(err), apierrors.IsTimeout(err):
|
||||
return provider.Wrap(provider.ErrTransient, op, "kubernetes", id, err)
|
||||
case apierrors.IsForbidden(err), apierrors.IsInvalid(err), apierrors.IsBadRequest(err), apierrors.IsUnauthorized(err):
|
||||
return provider.Wrap(provider.ErrPermanent, op, "kubernetes", id, err)
|
||||
default:
|
||||
return provider.Wrap(provider.ErrTransient, op, "kubernetes", id, err)
|
||||
}
|
||||
}
|
||||
245
internal/provider/kubernetes/kubernetes_test.go
Normal file
245
internal/provider/kubernetes/kubernetes_test.go
Normal file
@@ -0,0 +1,245 @@
|
||||
package kubernetes
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
"k8s.io/apimachinery/pkg/runtime/schema"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
func newTestProvider(objs ...runtime.Object) *Provider {
|
||||
scheme := runtime.NewScheme()
|
||||
if err := corev1.AddToScheme(scheme); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
c := fake.NewClientBuilder().WithScheme(scheme).WithRuntimeObjects(objs...).Build()
|
||||
return newWithClient(c, provider.ProviderConfig{Name: "kubernetes"})
|
||||
}
|
||||
|
||||
func testPod(namespace, name, uid string, phase corev1.PodPhase, ip string) *corev1.Pod {
|
||||
return &corev1.Pod{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: name,
|
||||
Namespace: namespace,
|
||||
Labels: map[string]string{
|
||||
provider.LabelManaged: provider.LabelManagedYes,
|
||||
provider.LabelUID: uid,
|
||||
},
|
||||
},
|
||||
Status: corev1.PodStatus{Phase: phase, PodIP: ip},
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Create_buildsPodAndReturnsProviderID(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider()
|
||||
id, err := p.Create(context.Background(), provider.CreateRequest{
|
||||
Name: "proxy-abc", UID: "uid-1", Namespace: "crawl", Port: 3128,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Create() error = %v", err)
|
||||
}
|
||||
if id != "crawl/proxy-abc" {
|
||||
t.Errorf("Create() id = %q, want %q", id, "crawl/proxy-abc")
|
||||
}
|
||||
|
||||
var pod corev1.Pod
|
||||
key := types.NamespacedName{Namespace: "crawl", Name: "proxy-abc"}
|
||||
if err := p.client.Get(context.Background(), key, &pod); err != nil {
|
||||
t.Fatalf("expected the Pod to exist: %v", err)
|
||||
}
|
||||
if pod.Labels[provider.LabelUID] != "uid-1" {
|
||||
t.Errorf("pod UID label = %q, want %q", pod.Labels[provider.LabelUID], "uid-1")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Create_idempotent(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider()
|
||||
ctx := context.Background()
|
||||
req := provider.CreateRequest{Name: "proxy-dup", UID: "uid-2", Namespace: "crawl", Port: 3128}
|
||||
|
||||
id1, err := p.Create(ctx, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Create() #1 error = %v", err)
|
||||
}
|
||||
id2, err := p.Create(ctx, req)
|
||||
if err != nil {
|
||||
t.Fatalf("Create() #2 error = %v, want nil (AlreadyExists must be absorbed)", err)
|
||||
}
|
||||
if id1 != id2 {
|
||||
t.Errorf("Create() not idempotent: %q != %q", id1, id2)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Get_stateMapping(t *testing.T) {
|
||||
t.Parallel()
|
||||
tests := []struct {
|
||||
name string
|
||||
phase corev1.PodPhase
|
||||
ip string
|
||||
wantState provider.InstanceState
|
||||
wantIP string
|
||||
}{
|
||||
{"pending", corev1.PodPending, "", provider.StateProvisioning, ""},
|
||||
{"running with IP", corev1.PodRunning, "10.0.0.5", provider.StateRunning, "10.0.0.5"},
|
||||
{"running without IP yet", corev1.PodRunning, "", provider.StateProvisioning, ""},
|
||||
{"succeeded", corev1.PodSucceeded, "10.0.0.5", provider.StateTerminated, ""},
|
||||
{"failed", corev1.PodFailed, "10.0.0.5", provider.StateTerminated, ""},
|
||||
{"unknown", corev1.PodUnknown, "10.0.0.5", provider.StateTerminated, ""},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
pod := testPod("crawl", "proxy-"+tc.name, "uid", tc.phase, tc.ip)
|
||||
p := newTestProvider(pod)
|
||||
inst, err := p.Get(context.Background(), "crawl/"+pod.Name)
|
||||
if err != nil {
|
||||
t.Fatalf("Get() error = %v", err)
|
||||
}
|
||||
if inst.State != tc.wantState {
|
||||
t.Errorf("State = %v, want %v", inst.State, tc.wantState)
|
||||
}
|
||||
if inst.IP != tc.wantIP {
|
||||
t.Errorf("IP = %q, want %q", inst.IP, tc.wantIP)
|
||||
}
|
||||
if inst.UID != "uid" {
|
||||
t.Errorf("UID = %q, want %q", inst.UID, "uid")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Get_notFound(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider()
|
||||
_, err := p.Get(context.Background(), "crawl/does-not-exist")
|
||||
if !errors.Is(err, provider.ErrNotFound) {
|
||||
t.Errorf("Get() error = %v, want ErrNotFound", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Get_malformedProviderID(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider()
|
||||
_, err := p.Get(context.Background(), "no-namespace-here")
|
||||
if err == nil {
|
||||
t.Fatal("Get() error = nil, want error for a providerID with no namespace")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Delete_idempotent(t *testing.T) {
|
||||
t.Parallel()
|
||||
pod := testPod("crawl", "proxy-del", "uid", corev1.PodRunning, "10.0.0.5")
|
||||
p := newTestProvider(pod)
|
||||
ctx := context.Background()
|
||||
|
||||
if err := p.Delete(ctx, "crawl/proxy-del"); err != nil {
|
||||
t.Fatalf("Delete() error = %v", err)
|
||||
}
|
||||
if err := p.Delete(ctx, "crawl/proxy-del"); err != nil {
|
||||
t.Errorf("Delete() (repeat) error = %v, want nil", err)
|
||||
}
|
||||
if err := p.Delete(ctx, "crawl/never-existed"); err != nil {
|
||||
t.Errorf("Delete() on unknown ID error = %v, want nil", err)
|
||||
}
|
||||
|
||||
if _, err := p.Get(ctx, "crawl/proxy-del"); !errors.Is(err, provider.ErrNotFound) {
|
||||
t.Errorf("Get() after Delete() error = %v, want ErrNotFound", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_Delete_malformedProviderID(t *testing.T) {
|
||||
t.Parallel()
|
||||
p := newTestProvider()
|
||||
if err := p.Delete(context.Background(), "no-namespace-here"); err == nil {
|
||||
t.Fatal("Delete() error = nil, want error for a providerID with no namespace")
|
||||
}
|
||||
}
|
||||
|
||||
func TestProvider_ListByTag_filtersByLabelAcrossNamespaces(t *testing.T) {
|
||||
t.Parallel()
|
||||
managed1 := testPod("crawl", "proxy-a", "uid-a", corev1.PodRunning, "10.0.0.1")
|
||||
managed2 := testPod("other-ns", "proxy-b", "uid-b", corev1.PodPending, "")
|
||||
unmanaged := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "not-ours", Namespace: "crawl"}}
|
||||
|
||||
p := newTestProvider(managed1, managed2, unmanaged)
|
||||
instances, err := p.ListByTag(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("ListByTag() error = %v", err)
|
||||
}
|
||||
if len(instances) != 2 {
|
||||
t.Fatalf("len(instances) = %d, want 2 (unmanaged Pod must be excluded)", len(instances))
|
||||
}
|
||||
|
||||
byID := make(map[string]provider.Instance, len(instances))
|
||||
for _, inst := range instances {
|
||||
byID[inst.ID] = inst
|
||||
}
|
||||
a, ok := byID["crawl/proxy-a"]
|
||||
if !ok {
|
||||
t.Fatal(`instances missing "crawl/proxy-a"`)
|
||||
}
|
||||
if a.State != provider.StateRunning || a.IP != "10.0.0.1" {
|
||||
t.Errorf("crawl/proxy-a = %+v, want Running/10.0.0.1", a)
|
||||
}
|
||||
if _, ok := byID["other-ns/proxy-b"]; !ok {
|
||||
t.Fatal(`instances missing "other-ns/proxy-b" (ListByTag must not be namespace-scoped)`)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewWithClient_image(t *testing.T) {
|
||||
t.Parallel()
|
||||
scheme := runtime.NewScheme()
|
||||
_ = corev1.AddToScheme(scheme)
|
||||
c := fake.NewClientBuilder().WithScheme(scheme).Build()
|
||||
|
||||
def := newWithClient(c, provider.ProviderConfig{Name: "k8s"})
|
||||
if def.image != defaultImage {
|
||||
t.Errorf("default image = %q, want %q", def.image, defaultImage)
|
||||
}
|
||||
|
||||
custom := newWithClient(c, provider.ProviderConfig{
|
||||
Name: "k8s",
|
||||
Kubernetes: &provider.KubernetesConfig{Image: "myregistry/squid:custom"},
|
||||
})
|
||||
if custom.image != "myregistry/squid:custom" {
|
||||
t.Errorf("custom image = %q, want %q", custom.image, "myregistry/squid:custom")
|
||||
}
|
||||
}
|
||||
|
||||
func TestClassify(t *testing.T) {
|
||||
t.Parallel()
|
||||
gr := schema.GroupResource{Group: "", Resource: "pods"}
|
||||
tests := []struct {
|
||||
name string
|
||||
err error
|
||||
want error
|
||||
}{
|
||||
{"not found", apierrors.NewNotFound(gr, "x"), provider.ErrNotFound},
|
||||
{"too many requests", apierrors.NewTooManyRequests("slow down", 5), provider.ErrTransient},
|
||||
{"server timeout", apierrors.NewServerTimeout(gr, "create", 5), provider.ErrTransient},
|
||||
{"forbidden", apierrors.NewForbidden(gr, "x", errors.New("denied")), provider.ErrPermanent},
|
||||
{"bad request", apierrors.NewBadRequest("bad"), provider.ErrPermanent},
|
||||
{"unauthorized", apierrors.NewUnauthorized("no creds"), provider.ErrPermanent},
|
||||
{"unclassified", errors.New("boom"), provider.ErrTransient},
|
||||
}
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
got := classify("op", "id", tc.err)
|
||||
if !errors.Is(got, tc.want) {
|
||||
t.Errorf("classify(%v) = %v, want class %v", tc.err, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
67
internal/provider/kubernetes/pod.go
Normal file
67
internal/provider/kubernetes/pod.go
Normal file
@@ -0,0 +1,67 @@
|
||||
package kubernetes
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
const proxyContainerName = "squid"
|
||||
|
||||
// writeConfigAndExec is the container's entrypoint: write the config the
|
||||
// SQUID_CONF env var carries to disk, then exec squid against it. Avoids a
|
||||
// separate ConfigMap object per proxy instance — there's still only one
|
||||
// Kubernetes object (the Pod) to create, track, and clean up per proxy.
|
||||
const writeConfigAndExec = `printf '%s' "$SQUID_CONF" > /etc/squid/squid.conf && exec squid -N -f /etc/squid/squid.conf`
|
||||
|
||||
// buildPod constructs the Pod for a proxy instance. Pure and side-effect
|
||||
// free, so it's unit-tested directly without a cluster — the same pattern
|
||||
// the GCP provider's buildInsertRequest uses.
|
||||
func buildPod(image string, req provider.CreateRequest) *corev1.Pod {
|
||||
return &corev1.Pod{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: req.Name,
|
||||
Namespace: req.Namespace,
|
||||
Labels: map[string]string{
|
||||
provider.LabelManaged: provider.LabelManagedYes,
|
||||
provider.LabelUID: req.UID,
|
||||
},
|
||||
},
|
||||
Spec: corev1.PodSpec{
|
||||
RestartPolicy: corev1.RestartPolicyAlways,
|
||||
Containers: []corev1.Container{{
|
||||
Name: proxyContainerName,
|
||||
Image: image,
|
||||
Command: []string{"/bin/sh", "-c"},
|
||||
Args: []string{writeConfigAndExec},
|
||||
Env: []corev1.EnvVar{{
|
||||
Name: "SQUID_CONF",
|
||||
Value: squidConf(req.Port),
|
||||
}},
|
||||
Ports: []corev1.ContainerPort{{
|
||||
ContainerPort: req.Port,
|
||||
Protocol: corev1.ProtocolTCP,
|
||||
}},
|
||||
}},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// squidConf generates a minimal Squid config listening on port, permissive
|
||||
// enough to forward CONNECT and plain HTTP to any destination. Open by
|
||||
// design: this is a proxy for a private cluster, not internet-facing, and
|
||||
// installing/configuring proxy software is explicitly out of scope for
|
||||
// this operator's real (GCP) provider too — cloud-init is passed through
|
||||
// there, never interpreted. via/forwarded_for are turned off so the proxy
|
||||
// doesn't leak the Pod's identity to the origin.
|
||||
func squidConf(port int32) string {
|
||||
return fmt.Sprintf(`http_port %d
|
||||
acl all src 0.0.0.0/0
|
||||
http_access allow all
|
||||
via off
|
||||
forwarded_for off
|
||||
`, port)
|
||||
}
|
||||
91
internal/provider/kubernetes/pod_test.go
Normal file
91
internal/provider/kubernetes/pod_test.go
Normal file
@@ -0,0 +1,91 @@
|
||||
package kubernetes
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/internal/provider"
|
||||
)
|
||||
|
||||
func TestBuildPod(t *testing.T) {
|
||||
t.Parallel()
|
||||
req := provider.CreateRequest{
|
||||
Name: "proxy-abc123",
|
||||
UID: "uid-1",
|
||||
Namespace: "crawl",
|
||||
ProxyName: "proxy-eu-1",
|
||||
Port: 3128,
|
||||
}
|
||||
pod := buildPod("ubuntu/squid:6.6-24.04_edge", req)
|
||||
|
||||
if pod.Name != req.Name {
|
||||
t.Errorf("pod.Name = %q, want %q", pod.Name, req.Name)
|
||||
}
|
||||
if pod.Namespace != req.Namespace {
|
||||
t.Errorf("pod.Namespace = %q, want %q", pod.Namespace, req.Namespace)
|
||||
}
|
||||
if pod.Labels[provider.LabelManaged] != provider.LabelManagedYes {
|
||||
t.Errorf("labels[%s] = %q, want %q", provider.LabelManaged, pod.Labels[provider.LabelManaged], provider.LabelManagedYes)
|
||||
}
|
||||
if pod.Labels[provider.LabelUID] != req.UID {
|
||||
t.Errorf("labels[%s] = %q, want %q", provider.LabelUID, pod.Labels[provider.LabelUID], req.UID)
|
||||
}
|
||||
if pod.Spec.RestartPolicy != corev1.RestartPolicyAlways {
|
||||
t.Errorf("RestartPolicy = %v, want Always", pod.Spec.RestartPolicy)
|
||||
}
|
||||
if len(pod.Spec.Containers) != 1 {
|
||||
t.Fatalf("len(Containers) = %d, want 1", len(pod.Spec.Containers))
|
||||
}
|
||||
c := pod.Spec.Containers[0]
|
||||
if c.Image != "ubuntu/squid:6.6-24.04_edge" {
|
||||
t.Errorf("Image = %q, want ubuntu/squid:6.6-24.04_edge", c.Image)
|
||||
}
|
||||
if len(c.Ports) != 1 || c.Ports[0].ContainerPort != req.Port {
|
||||
t.Errorf("Ports = %+v, want a single entry on port %d", c.Ports, req.Port)
|
||||
}
|
||||
var confEnv string
|
||||
for _, e := range c.Env {
|
||||
if e.Name == "SQUID_CONF" {
|
||||
confEnv = e.Value
|
||||
}
|
||||
}
|
||||
if !strings.Contains(confEnv, "http_port 3128") {
|
||||
t.Errorf("SQUID_CONF env = %q, want it to contain %q", confEnv, "http_port 3128")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildPod_usesRequestPort(t *testing.T) {
|
||||
t.Parallel()
|
||||
req := provider.CreateRequest{Name: "proxy-x", UID: "uid-2", Namespace: "ns", Port: 8080}
|
||||
pod := buildPod("img", req)
|
||||
conf := envValue(t, pod, "SQUID_CONF")
|
||||
if !strings.Contains(conf, "http_port 8080") {
|
||||
t.Errorf("SQUID_CONF = %q, want it to contain %q", conf, "http_port 8080")
|
||||
}
|
||||
if pod.Spec.Containers[0].Ports[0].ContainerPort != 8080 {
|
||||
t.Errorf("ContainerPort = %d, want 8080", pod.Spec.Containers[0].Ports[0].ContainerPort)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSquidConf_permissive(t *testing.T) {
|
||||
t.Parallel()
|
||||
conf := squidConf(3128)
|
||||
for _, want := range []string{"http_port 3128", "http_access allow all", "via off", "forwarded_for off"} {
|
||||
if !strings.Contains(conf, want) {
|
||||
t.Errorf("squidConf() = %q, want it to contain %q", conf, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func envValue(t *testing.T, pod *corev1.Pod, name string) string {
|
||||
t.Helper()
|
||||
for _, e := range pod.Spec.Containers[0].Env {
|
||||
if e.Name == name {
|
||||
return e.Value
|
||||
}
|
||||
}
|
||||
t.Fatalf("env var %q not found on container", name)
|
||||
return ""
|
||||
}
|
||||
@@ -31,19 +31,14 @@ import (
|
||||
"gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/test/utils"
|
||||
)
|
||||
|
||||
var (
|
||||
// managerImage is the manager image to be built and loaded for testing.
|
||||
managerImage = "example.com/egress-proxies-operator:v0.0.1"
|
||||
// shouldCleanupCertManager tracks whether CertManager was installed by this suite.
|
||||
shouldCleanupCertManager = false
|
||||
)
|
||||
var managerImage = "example.com/egress-proxies-operator:v0.0.1"
|
||||
|
||||
// TestE2E runs the e2e test suite to validate the solution in an isolated environment.
|
||||
// The default setup requires Kind and CertManager.
|
||||
// TestE2E runs the e2e test suite to validate the solution in an isolated
|
||||
// environment. The default setup requires Kind.
|
||||
//
|
||||
// To enable kubectl kuberc (use custom kubectl configurations), set: KUBECTL_KUBERC=true
|
||||
// By default, kuberc is disabled to ensure consistent test behavior across different environments.
|
||||
// To skip CertManager installation, set: CERT_MANAGER_INSTALL_SKIP=true
|
||||
func TestE2E(t *testing.T) {
|
||||
RegisterFailHandler(Fail)
|
||||
_, _ = fmt.Fprintf(GinkgoWriter, "Starting egress-proxies-operator e2e test suite\n")
|
||||
@@ -56,18 +51,11 @@ var _ = BeforeSuite(func() {
|
||||
_, err := utils.Run(cmd)
|
||||
ExpectWithOffset(1, err).NotTo(HaveOccurred(), "Failed to build the manager image")
|
||||
|
||||
// TODO(user): If you want to change the e2e test vendor from Kind,
|
||||
// ensure the image is built and available, then remove the following block.
|
||||
By("loading the manager image on Kind")
|
||||
err = utils.LoadImageToKindClusterWithName(managerImage)
|
||||
ExpectWithOffset(1, err).NotTo(HaveOccurred(), "Failed to load the manager image into Kind")
|
||||
|
||||
configureKubectlKubeRC()
|
||||
setupCertManager()
|
||||
})
|
||||
|
||||
var _ = AfterSuite(func() {
|
||||
teardownCertManager()
|
||||
})
|
||||
|
||||
// Disable kubectl kuberc by default for test isolation.
|
||||
@@ -84,36 +72,3 @@ func configureKubectlKubeRC() {
|
||||
_, _ = fmt.Fprintf(GinkgoWriter, "kubectl kuberc enabled (KUBECTL_KUBERC=true)\n")
|
||||
}
|
||||
}
|
||||
|
||||
// setupCertManager installs CertManager if needed for webhook tests.
|
||||
// Skips installation if CERT_MANAGER_INSTALL_SKIP=true or if already present.
|
||||
func setupCertManager() {
|
||||
if os.Getenv("CERT_MANAGER_INSTALL_SKIP") == "true" {
|
||||
_, _ = fmt.Fprintf(GinkgoWriter, "Skipping CertManager installation (CERT_MANAGER_INSTALL_SKIP=true)\n")
|
||||
return
|
||||
}
|
||||
|
||||
By("checking if CertManager is already installed")
|
||||
if utils.IsCertManagerCRDsInstalled() {
|
||||
_, _ = fmt.Fprintf(GinkgoWriter, "CertManager is already installed. Skipping installation.\n")
|
||||
return
|
||||
}
|
||||
|
||||
// Mark for cleanup before installation to handle interruptions and partial installs.
|
||||
shouldCleanupCertManager = true
|
||||
|
||||
By("installing CertManager")
|
||||
Expect(utils.InstallCertManager()).To(Succeed(), "Failed to install CertManager")
|
||||
}
|
||||
|
||||
// teardownCertManager uninstalls CertManager if it was installed by setupCertManager.
|
||||
// This ensures we only remove what we installed.
|
||||
func teardownCertManager() {
|
||||
if !shouldCleanupCertManager {
|
||||
_, _ = fmt.Fprintf(GinkgoWriter, "Skipping CertManager cleanup (not installed by this suite)\n")
|
||||
return
|
||||
}
|
||||
|
||||
By("uninstalling CertManager")
|
||||
utils.UninstallCertManager()
|
||||
}
|
||||
|
||||
@@ -17,8 +17,6 @@ limitations under the License.
|
||||
package utils
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
@@ -28,17 +26,10 @@ import (
|
||||
)
|
||||
|
||||
const (
|
||||
certmanagerVersion = "v1.20.2"
|
||||
certmanagerURLTmpl = "https://github.com/cert-manager/cert-manager/releases/download/%s/cert-manager.yaml"
|
||||
|
||||
defaultKindBinary = "kind"
|
||||
defaultKindCluster = "kind"
|
||||
)
|
||||
|
||||
func warnError(err error) {
|
||||
_, _ = fmt.Fprintf(GinkgoWriter, "warning: %v\n", err)
|
||||
}
|
||||
|
||||
// Run executes the provided command within this context
|
||||
func Run(cmd *exec.Cmd) (string, error) {
|
||||
dir, _ := GetProjectDir()
|
||||
@@ -59,80 +50,6 @@ func Run(cmd *exec.Cmd) (string, error) {
|
||||
return string(output), nil
|
||||
}
|
||||
|
||||
// UninstallCertManager uninstalls the cert manager
|
||||
func UninstallCertManager() {
|
||||
url := fmt.Sprintf(certmanagerURLTmpl, certmanagerVersion)
|
||||
cmd := exec.Command("kubectl", "delete", "-f", url)
|
||||
if _, err := Run(cmd); err != nil {
|
||||
warnError(err)
|
||||
}
|
||||
|
||||
// Delete leftover leases in kube-system (not cleaned by default)
|
||||
kubeSystemLeases := []string{
|
||||
"cert-manager-cainjector-leader-election",
|
||||
"cert-manager-controller",
|
||||
}
|
||||
for _, lease := range kubeSystemLeases {
|
||||
cmd = exec.Command("kubectl", "delete", "lease", lease,
|
||||
"-n", "kube-system", "--ignore-not-found", "--force", "--grace-period=0")
|
||||
if _, err := Run(cmd); err != nil {
|
||||
warnError(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// InstallCertManager installs the cert manager bundle.
|
||||
func InstallCertManager() error {
|
||||
url := fmt.Sprintf(certmanagerURLTmpl, certmanagerVersion)
|
||||
cmd := exec.Command("kubectl", "apply", "-f", url)
|
||||
if _, err := Run(cmd); err != nil {
|
||||
return err
|
||||
}
|
||||
// Wait for cert-manager-webhook to be ready, which can take time if cert-manager
|
||||
// was re-installed after uninstalling on a cluster.
|
||||
cmd = exec.Command("kubectl", "wait", "deployment.apps/cert-manager-webhook",
|
||||
"--for", "condition=Available",
|
||||
"--namespace", "cert-manager",
|
||||
"--timeout", "5m",
|
||||
)
|
||||
|
||||
_, err := Run(cmd)
|
||||
return err
|
||||
}
|
||||
|
||||
// IsCertManagerCRDsInstalled checks if any Cert Manager CRDs are installed
|
||||
// by verifying the existence of key CRDs related to Cert Manager.
|
||||
func IsCertManagerCRDsInstalled() bool {
|
||||
// List of common Cert Manager CRDs
|
||||
certManagerCRDs := []string{
|
||||
"certificates.cert-manager.io",
|
||||
"issuers.cert-manager.io",
|
||||
"clusterissuers.cert-manager.io",
|
||||
"certificaterequests.cert-manager.io",
|
||||
"orders.acme.cert-manager.io",
|
||||
"challenges.acme.cert-manager.io",
|
||||
}
|
||||
|
||||
// Execute the kubectl command to get all CRDs
|
||||
cmd := exec.Command("kubectl", "get", "crds")
|
||||
output, err := Run(cmd)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
|
||||
// Check if any of the Cert Manager CRDs are present
|
||||
crdList := GetNonEmptyLines(output)
|
||||
for _, crd := range certManagerCRDs {
|
||||
for _, line := range crdList {
|
||||
if strings.Contains(line, crd) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false
|
||||
}
|
||||
|
||||
// LoadImageToKindClusterWithName loads a local docker image to the kind cluster
|
||||
func LoadImageToKindClusterWithName(name string) error {
|
||||
cluster := defaultKindCluster
|
||||
@@ -172,55 +89,3 @@ func GetProjectDir() (string, error) {
|
||||
wd = strings.ReplaceAll(wd, "/test/e2e", "")
|
||||
return wd, nil
|
||||
}
|
||||
|
||||
// UncommentCode searches for target in the file and remove the comment prefix
|
||||
// of the target content. The target content may span multiple lines.
|
||||
func UncommentCode(filename, target, prefix string) error {
|
||||
// false positive
|
||||
// nolint:gosec
|
||||
content, err := os.ReadFile(filename)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to read file %q: %w", filename, err)
|
||||
}
|
||||
strContent := string(content)
|
||||
|
||||
idx := strings.Index(strContent, target)
|
||||
if idx < 0 {
|
||||
return fmt.Errorf("unable to find the code %q to be uncommented", target)
|
||||
}
|
||||
|
||||
out := new(bytes.Buffer)
|
||||
_, err = out.Write(content[:idx])
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to write to output: %w", err)
|
||||
}
|
||||
|
||||
scanner := bufio.NewScanner(bytes.NewBufferString(target))
|
||||
if !scanner.Scan() {
|
||||
return nil
|
||||
}
|
||||
for {
|
||||
if _, err = out.WriteString(strings.TrimPrefix(scanner.Text(), prefix)); err != nil {
|
||||
return fmt.Errorf("failed to write to output: %w", err)
|
||||
}
|
||||
// Avoid writing a newline in case the previous line was the last in target.
|
||||
if !scanner.Scan() {
|
||||
break
|
||||
}
|
||||
if _, err = out.WriteString("\n"); err != nil {
|
||||
return fmt.Errorf("failed to write to output: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
if _, err = out.Write(content[idx+len(target):]); err != nil {
|
||||
return fmt.Errorf("failed to write to output: %w", err)
|
||||
}
|
||||
|
||||
// false positive
|
||||
// nolint:gosec
|
||||
if err = os.WriteFile(filename, out.Bytes(), 0644); err != nil {
|
||||
return fmt.Errorf("failed to write file %q: %w", filename, err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user