8 Commits

Author SHA1 Message Date
d2c317344e Merge pull request 'Add Gitea Actions image-build workflow' (#2) from feat/gitea-build-workflow into main
All checks were successful
Build and Push / check (push) Successful in 36s
Build and Push / build (push) Successful in 3m21s
Reviewed-on: #2
2026-08-11 19:47:42 +02:00
9230b1213c Document Gitea CI and required secrets in README
Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-11 19:46:55 +02:00
57e3ea22cf Record MR creation in plan execution summary
Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-11 19:41:57 +02:00
95c487415b Add Gitea Actions image-build workflow
Distilled from the house pattern across sibling projects: tag push +
workflow_dispatch triggers, REGISTRY_TOKEN login, raw docker build/push
to gitea.home.hrajfrisbee.cz. Adds a lightweight test gate, an immutable
sha-<12> tag, :latest only on real tag pushes, and a concurrency group.

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-11 19:39:23 +02:00
849ec1083e Add plan: distilled Gitea Actions image-build workflow
Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-11 19:37:42 +02:00
f7000f7514 Merge pull request 'proxy-operator: Kubernetes operator for crawling-proxy fleets' (#1) from feat/proxy-operator into main
Reviewed-on: #1
2026-08-11 19:21:14 +02:00
19d6a8dfba Add GCP deployment docs, PR review notes, and Claude tooling updates
docs/gcp-in-specific-project.md: SA + firewall setup for the egress-proxy
project, in-kube secret, and apply-ready ConfigMap/Deployment/Proxy
manifests (Ubuntu image — debian-cloud lacks cloud-init).
docs/gcp-vm-validation.md: end-to-end GCP VM validation walkthrough.
docs/reviews/: proxy-operator PR review notes from 2026-08-10.
.claude/: operator-reviewer agent, accumulated permission allowlist.
.gitignore: never commit sa_key.json (live SA key stays untracked).

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-11 19:17:58 +02:00
420c3509b0 Wire logging: drop auth token-exchange records, elide huge payload fields
The option.WithLogger logger also reaches cloud.google.com/go/auth,
which logged its token exchange at Debug — JWT assertion and bearer
token included. wireLogger now allowlists only the compute client's
api request/response records at Debug (fail-closed for future SDK
additions); Warn/Error pass through. String fields over 1KiB (e.g.
Shielded-VM UEFI dbx blobs) are elided recursively by default; the new
--gcp-wire-log-full-payloads flag restores verbatim payloads.

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-11 19:02:28 +02:00
16 changed files with 1123 additions and 23 deletions

View File

@@ -0,0 +1,15 @@
---
name: operator-reviewer
description: Reviews Kubernetes operator PRs for controller-runtime correctness, reconcile semantics, and API design
tools: Read, Grep, Glob, Bash
---
You are a senior reviewer specializing in Kubernetes operators.
Review with focus on:
- Reconcile idempotency and requeue behavior; no state assumptions between reconciles
- Informer cache reads vs direct API reads; stale-cache races
- Finalizer handling, deletion flow, orphaned resources
- CRD schema evolution, conversion webhooks, status subresource / conditions conventions
- RBAC minimality vs what the controller actually touches
- Leader election, watch predicates, event filtering for churn reduction
- Go: context propagation, error wrapping, client.Object handling
Output: findings ranked by severity, with file:line refs. No praise padding.

View File

@@ -71,7 +71,19 @@
"Bash(kind load *)", "Bash(kind load *)",
"Bash(make deploy *)", "Bash(make deploy *)",
"Bash(kubectl -n egress-proxies-operator-system rollout status deploy/egress-proxies-operator-controller-manager --timeout=120s)", "Bash(kubectl -n egress-proxies-operator-system rollout status deploy/egress-proxies-operator-controller-manager --timeout=120s)",
"Bash(kubectl -n egress-proxies-operator-system rollout restart deploy/egress-proxies-operator-controller-manager)" "Bash(kubectl -n egress-proxies-operator-system rollout restart deploy/egress-proxies-operator-controller-manager)",
"Bash(git -C /Users/jan.novak/srv/go/egress-proxies-operator add docs/plans/2026-08-11-1742-gcp-provider-verbose-logging.md)",
"Bash(git -C /Users/jan.novak/srv/go/egress-proxies-operator commit -m 'Add plan: verbose V-level logging in the GCP provider *)",
"Bash(echo \"exit: $?\")",
"Bash(echo \"tests exit: $?\")",
"Bash(./bin/manager --version)",
"Bash(./bin/manager-stamped --version)",
"Bash(./bin/manager-pkg --version)",
"Bash(docker run *)",
"Bash(kubectl -n egress-proxies-operator-system get pods -o wide)",
"Bash(kubectl -n egress-proxies-operator-system get deploy egress-proxies-operator-controller-manager -o jsonpath='{.spec.template.spec.containers[0].args}')",
"Bash(kubectl -n egress-proxies-operator-system logs deploy/egress-proxies-operator-controller-manager)",
"Bash(python3 -c \"import json; d=json.load\\(open\\('docs/deploy/sa_key.json'\\)\\); print\\(d.get\\('type'\\), d.get\\('client_email'\\)\\)\")"
], ],
"additionalDirectories": [ "additionalDirectories": [
"/Users/jan.novak/srv/go/egress-proxies-operator/.claude", "/Users/jan.novak/srv/go/egress-proxies-operator/.claude",

View File

@@ -0,0 +1,76 @@
name: Build and Push
on:
workflow_dispatch:
inputs:
tag:
description: 'Image tag'
required: true
default: 'latest'
push:
tags:
- '*'
concurrency:
group: build-${{ github.ref }}
cancel-in-progress: true
jobs:
check:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-go@v5
with:
go-version-file: go.mod
cache: true
- name: Vet
run: go vet ./...
- name: Build
run: go build ./...
- name: Test (short)
run: go test -short ./...
build:
needs: check
runs-on: ubuntu-latest
permissions:
contents: read
packages: write
steps:
- uses: actions/checkout@v4
- name: Compute image tags
id: meta
run: |
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
TAG="${{ inputs.tag }}"
else
TAG="${{ github.ref_name }}"
fi
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
echo "sha=sha-$(echo '${{ github.sha }}' | cut -c1-12)" >> "$GITHUB_OUTPUT"
- name: Login to Gitea registry
run: echo "${{ secrets.REGISTRY_TOKEN }}" | docker login -u ${{ github.actor }} --password-stdin gitea.home.hrajfrisbee.cz
- name: Build and push
run: |
IMAGE=gitea.home.hrajfrisbee.cz/${{ github.repository }}
docker build \
--build-arg GIT_COMMIT=$(echo '${{ github.sha }}' | cut -c1-12) \
--label org.opencontainers.image.source=https://gitea.home.hrajfrisbee.cz/${{ github.repository }} \
--label org.opencontainers.image.created=$(date -u +%Y-%m-%dT%H:%M:%SZ) \
-t "$IMAGE:${{ steps.meta.outputs.tag }}" \
-t "$IMAGE:${{ steps.meta.outputs.sha }}" \
.
docker push "$IMAGE:${{ steps.meta.outputs.tag }}"
docker push "$IMAGE:${{ steps.meta.outputs.sha }}"
# Only real tag pushes move :latest — an ad-hoc dispatch of an old ref must not clobber it.
- name: Push latest (tag builds only)
if: github.event_name == 'push'
run: |
IMAGE=gitea.home.hrajfrisbee.cz/${{ github.repository }}
docker tag "$IMAGE:${{ steps.meta.outputs.tag }}" "$IMAGE:latest"
docker push "$IMAGE:latest"

3
.gitignore vendored
View File

@@ -28,3 +28,6 @@ go.work
# Kubeconfig might contain secrets # Kubeconfig might contain secrets
*.kubeconfig *.kubeconfig
# GCP service-account keys (created per docs/gcp-in-specific-project.md)
sa_key.json

View File

@@ -199,8 +199,18 @@ Always append a `Co-Authored-By` trailer to indicate AI assistance:
Co-Authored-By: Claude <noreply@anthropic.com> Co-Authored-By: Claude <noreply@anthropic.com>
TODO: no `.gitea/workflows/` CI pipeline exists yet — add a CI/CD subsection here once ### CI/CD
one is set up.
`.gitea/workflows/build.yaml` builds the manager image and pushes it to the Gitea
registry. Triggers: any tag push, or manual `workflow_dispatch` with a `tag` input.
A lightweight `check` job (`go vet` / `go build` / `go test -short`) gates the build.
- Images: `gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator:<tag>` plus an
immutable `sha-<12-char-commit>` tag on every build; `:latest` moves only on real
tag pushes, never on manual dispatch.
- Requires the `REGISTRY_TOKEN` repo secret (Gitea PAT with `write:package`),
same convention as the other projects on this Gitea instance.
- The commit is baked into the binary via the `GIT_COMMIT` build arg (see Dockerfile).
## Gotchas ## Gotchas

View File

@@ -167,6 +167,40 @@ cloud.google.com/go/compute v1.65.0. envtest uses the 1.36.2 binary
bundle (the latest 1.36 patch with published binaries — do not "fix" the bundle (the latest 1.36 patch with published binaries — do not "fix" the
Makefile's derived version to 1.36.3, which has none). Makefile's derived version to 1.36.3, which has none).
## Gitea CI
[.gitea/workflows/build.yaml](.gitea/workflows/build.yaml) builds the
manager image and pushes it to this Gitea instance's container registry.
It runs on **any tag push** or manually via **Run workflow** (with a `tag`
input) — never on branch pushes. A lightweight `check` job (`go vet`,
`go build`, `go test -short`) gates the build.
Every build pushes two tags to
`gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator`:
- the human tag (the git tag, or the dispatch input), and
- an immutable `sha-<12-char-commit>` tag — pin deployments to this one.
`:latest` is additionally updated on real tag pushes only, so a manual
dispatch of an old ref can never clobber it. The commit is baked into the
binary (`internal/version.Commit`) via the `GIT_COMMIT` build arg.
### Mandatory Gitea secrets
Set under **Settings → Actions → Secrets** in this repo:
| Secret | Required by | What it is |
| ---------------- | ----------------------------- | ---------------------------------------- |
| `REGISTRY_TOKEN` | `build.yaml` (registry login) | Gitea PAT with the `write:package` scope |
The token is paired with `${{ github.actor }}` as the username, so it
must belong to the user triggering the workflow — same convention as the
other projects on this instance.
Without `REGISTRY_TOKEN` the `check` job still passes but the build job
fails at the `docker login` step. No other secrets are needed — the
workflow does not deploy anywhere.
## Development ## Development
```sh ```sh

View File

@@ -92,6 +92,7 @@ func main() {
var gcAllowNamespaced bool var gcAllowNamespaced bool
var leaseCooldown, maxLeaseTTL time.Duration var leaseCooldown, maxLeaseTTL time.Duration
var showVersion bool var showVersion bool
var gcpWireFullPayloads bool
flag.StringVar(&metricsAddr, "metrics-bind-address", "0", "The address the metrics endpoint binds to. "+ flag.StringVar(&metricsAddr, "metrics-bind-address", "0", "The address the metrics endpoint binds to. "+
"Use :8443 for HTTPS or :8080 for HTTP, or leave as 0 to disable the metrics service.") "Use :8443 for HTTPS or :8080 for HTTP, or leave as 0 to disable the metrics service.")
@@ -130,6 +131,8 @@ func main() {
"Maximum lease TTL a client may request.") "Maximum lease TTL a client may request.")
flag.BoolVar(&showVersion, "version", false, flag.BoolVar(&showVersion, "version", false,
"Print the commit the binary was built from and exit.") "Print the commit the binary was built from and exit.")
flag.BoolVar(&gcpWireFullPayloads, "gcp-wire-log-full-payloads", false,
"Log GCP V(5) wire payloads verbatim instead of eliding fields larger than 1KiB.")
opts := zap.Options{ opts := zap.Options{
Development: true, Development: true,
@@ -160,7 +163,9 @@ func main() {
} }
providers, err := registry.Build(ctx, cfg, map[string]registry.Constructor{ providers, err := registry.Build(ctx, cfg, map[string]registry.Constructor{
"kubernetes": kubernetes.New, "kubernetes": kubernetes.New,
"gcp": gcp.New, "gcp": func(ctx context.Context, pc provider.ProviderConfig) (provider.Provider, error) {
return gcp.NewWithWireOptions(ctx, pc, gcp.WireLogOptions{FullPayloads: gcpWireFullPayloads})
},
}) })
if err != nil { if err != nil {
setupLog.Error(err, "Failed to build providers") setupLog.Error(err, "Failed to build providers")

View File

@@ -0,0 +1,255 @@
## Project egress-proxy
```bash
PROJECT_ID=egress-proxy
# 1. Create the service account
gcloud iam service-accounts create proxy-operator \
--project ${PROJECT_ID} \
--display-name "egress-proxies-operator"
# Output:
# Created service account [proxy-operator].
# Service account email: proxy-operator@egress-proxy.iam.gserviceaccount.com
# 2. Grant compute.instanceAdmin.v1 on the project
gcloud projects add-iam-policy-binding ${PROJECT_ID} \
--member "serviceAccount:proxy-operator@${PROJECT_ID}.iam.gserviceaccount.com" \
--role roles/compute.instanceAdmin.v1
# Output:
# ---------
# Updated IAM policy for project [egress-proxy].
# bindings:
# - members:
# - serviceAccount:proxy-operator@egress-proxy.iam.gserviceaccount.com
# role: roles/compute.instanceAdmin.v1
# - members:
# - serviceAccount:541231138892@cloudservices.gserviceaccount.com
# role: roles/compute.instanceGroupManagerServiceAgent
# - members:
# - serviceAccount:service-541231138892@compute-system.iam.gserviceaccount.com
# role: roles/compute.serviceAgent
# - members:
# - user:admin@fujultimate.cz
# role: roles/owner
# etag: BwZYuFXko24=
# version: 1
# 3. Create the JSON key (this is what goes into the Secret)
SA_KEY_PATH=sa_key.json
gcloud iam service-accounts keys create $SA_KEY_PATH \
--iam-account proxy-operator@${PROJECT_ID}.iam.gserviceaccount.com
# output:
# created key [fdff85174a8e80bbd684e76c4d9fe28e2f4b2ddf] of type [json] as [sa_key.json] for [proxy-operator@egress-proxy.iam.gserviceaccount.com]
# 4. A **firewall rule**: created VMs get network tag `proxy-operator` (the
# default; configurable as `gcp.networkTag`), an ephemeral external IP,
# and Squid listening on 3128.
gcloud compute firewall-rules create allow-proxy-operator \
--project $PROJECT_ID \
--network default \
--allow tcp:3128 \
--target-tags proxy-operator \
--source-ranges 94.230.145.216/32
```
## Phase 2 - resources in kube
```bash
SA_KEY_PATH=sa_key.json
kubectl -n egress-proxies-operator-system create secret generic gcp-credentials \
--from-file=key.json=$SA_KEY_PATH
```
## Appendix - full manifests
```bash
# crawl CR
kubectl apply -f - <<'EOF'
apiVersion: crawl.example.com/v1alpha1
kind: Proxy
metadata:
name: proxy-gcp-sample
spec:
mode: Managed
provider: gcp-eu # must match a provider NAME in providers.yaml
placement:
zone: europe-west1-b
machineType: e2-micro
# debian-cloud images have no cloud-init, so spec.cloudInit (passed as
# user-data metadata) would be silently ignored there. Ubuntu images do.
image: projects/ubuntu-os-cloud/global/images/family/ubuntu-2404-lts-amd64
port: 3128
cloudInit:
inline: |
#cloud-config
package_update: true
packages:
- squid
write_files:
- path: /etc/squid/conf.d/proxy-operator.conf
content: |
http_access allow all
via off
forwarded_for off
runcmd:
- systemctl restart squid
attributes:
geo: eu
purpose: crawl
EOF
# configmap
kubectl apply -f - <<'EOF'
apiVersion: v1
data:
providers.yaml: |
providers:
- name: kubernetes
type: kubernetes
- name: gcp-eu # spec.provider on a Proxy refers to this NAME, not the type
type: gcp
gcp:
project: egress-proxy
# network: default # these three default as shown
# networkTag: proxy-operator
# diskSizeGb: 10
kind: ConfigMap
metadata:
labels:
app.kubernetes.io/managed-by: kustomize
app.kubernetes.io/name: egress-proxies-operator
name: egress-proxies-operator-providers-config
namespace: egress-proxies-operator-system
EOF
# operator deployment
kubectl apply -f - <<'EOF'
apiVersion: apps/v1
kind: Deployment
metadata:
annotations:
deployment.kubernetes.io/revision: "2"
labels:
app.kubernetes.io/managed-by: kustomize
app.kubernetes.io/name: egress-proxies-operator
control-plane: controller-manager
name: egress-proxies-operator-controller-manager
namespace: egress-proxies-operator-system
spec:
progressDeadlineSeconds: 600
replicas: 1
revisionHistoryLimit: 10
selector:
matchLabels:
app.kubernetes.io/name: egress-proxies-operator
control-plane: controller-manager
strategy:
rollingUpdate:
maxSurge: 25%
maxUnavailable: 25%
type: RollingUpdate
template:
metadata:
annotations:
kubectl.kubernetes.io/default-container: manager
labels:
app.kubernetes.io/name: egress-proxies-operator
control-plane: controller-manager
spec:
containers:
- args:
- --metrics-bind-address=:8443
- --leader-elect
- --health-probe-bind-address=:8081
- --providers-config=/etc/proxy-operator/providers.yaml
command:
- /manager
env:
- name: DISCOVERY_TOKEN
valueFrom:
secretKeyRef:
key: token
name: discovery-token
optional: true
- name: GOOGLE_APPLICATION_CREDENTIALS
value: /var/secrets/gcp/key.json
image: egress-proxies-operator:dev
imagePullPolicy: IfNotPresent
livenessProbe:
failureThreshold: 3
httpGet:
path: /healthz
port: 8081
scheme: HTTP
initialDelaySeconds: 15
periodSeconds: 20
successThreshold: 1
timeoutSeconds: 1
name: manager
ports:
- containerPort: 8081
name: health
protocol: TCP
- containerPort: 8090
name: discovery
protocol: TCP
readinessProbe:
failureThreshold: 3
httpGet:
path: /readyz
port: 8081
scheme: HTTP
initialDelaySeconds: 5
periodSeconds: 10
successThreshold: 1
timeoutSeconds: 1
resources:
limits:
cpu: 500m
memory: 128Mi
requests:
cpu: 10m
memory: 64Mi
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop:
- ALL
readOnlyRootFilesystem: true
terminationMessagePath: /dev/termination-log
terminationMessagePolicy: File
volumeMounts:
- mountPath: /etc/proxy-operator
name: providers-config
readOnly: true
- mountPath: /var/secrets/gcp
name: gcp-credentials
readOnly: true
dnsPolicy: ClusterFirst
restartPolicy: Always
schedulerName: default-scheduler
securityContext:
runAsNonRoot: true
seccompProfile:
type: RuntimeDefault
serviceAccount: egress-proxies-operator-controller-manager
serviceAccountName: egress-proxies-operator-controller-manager
terminationGracePeriodSeconds: 10
volumes:
- configMap:
defaultMode: 420
name: egress-proxies-operator-providers-config
name: providers-config
- name: gcp-credentials
secret:
defaultMode: 420
secretName: gcp-credentials
EOF
```

158
docs/gcp-vm-validation.md Normal file
View File

@@ -0,0 +1,158 @@
# Validating real VM creation on GCP
Recipe for wiring the GCP provider into a live cluster and watching a
`Proxy` CR create a real Compute Engine VM. Angle brackets mark values you
supply: `<PROJECT_ID>`, `<SA_KEY_PATH>`, `<ZONE>`, `<CLUSTER_EGRESS_IP>`,
`<REGISTRY_IMAGE>`.
The one important fact up front: **the operator takes no GCP credentials
through its own config.** The client is built with Application Default
Credentials (`internal/provider/gcp/gcp.go`, `New()`); there is no
key-file field in the providers config. The only secret to prepare is a
service-account JSON key, injected via the standard
`GOOGLE_APPLICATION_CREDENTIALS` mechanism. On GKE you would use workload
identity instead and skip the key entirely.
## 1. GCP-side prerequisites (prepared outside the cluster)
1. A project — `<PROJECT_ID>` — with the **Compute Engine API enabled**.
2. A **service account** with `roles/compute.instanceAdmin.v1` on the
project. The operator only calls instances
`Insert`/`Get`/`Delete`/`AggregatedList` and does not attach a service
account to the VMs it creates, so no `iam.serviceAccountUser` is
needed.
3. A **JSON key** for that service account, saved at `<SA_KEY_PATH>`.
4. A **firewall rule**: created VMs get network tag `proxy-operator` (the
default; configurable as `gcp.networkTag`), an ephemeral external IP,
and Squid listening on 3128.
```sh
gcloud compute firewall-rules create allow-proxy-operator \
--project <PROJECT_ID> \
--network default \
--allow tcp:3128 \
--target-tags proxy-operator \
--source-ranges <CLUSTER_EGRESS_IP>/32
```
The source range must cover the cluster's egress IP — the operator's
CONNECT health probes originate there, and without the rule the Proxy
hangs at `Running`/unhealthy instead of reaching `Ready`. ⚠️ The
sample cloud-init configures `http_access allow all`, so on a public
IP this is an open proxy — keep the source ranges tight.
## 2. Create the credentials Secret
Namespace is `egress-proxies-operator-system` after kustomize prefixing:
```sh
kubectl -n egress-proxies-operator-system create secret generic gcp-credentials \
--from-file=key.json=<SA_KEY_PATH>
```
## 3. Add a GCP entry to the providers ConfigMap
Edit `config/manager/providers_config.yaml` (mounted at
`/etc/proxy-operator/providers.yaml`):
```yaml
providers:
- name: kubernetes
type: kubernetes
- name: gcp-eu # spec.provider on a Proxy refers to this NAME, not the type
type: gcp
gcp:
project: <PROJECT_ID>
# network: default # these three default as shown
# networkTag: proxy-operator
# diskSizeGb: 10
```
The config is validated fail-fast at startup — a typo shows up
immediately in the manager log, not on first use.
## 4. Mount the Secret and point ADC at it
In `config/manager/manager.yaml`, add to the manager container:
```yaml
env:
- name: GOOGLE_APPLICATION_CREDENTIALS
value: /var/secrets/gcp/key.json
volumeMounts:
- name: gcp-credentials
mountPath: /var/secrets/gcp
readOnly: true
volumes:
- name: gcp-credentials
secret:
secretName: gcp-credentials
```
(`volumeMounts` merges into the existing container list; `volumes` into
the existing pod-level list.)
## 5. Deploy and create the Proxy
```sh
make deploy IMG=<REGISTRY_IMAGE>
```
`config/samples/proxy_gcp.yaml` is usable as-is once `spec.provider`
matches the name from step 3. All three placement fields are mandatory
for GCP — a missing one sets the Proxy to `Failed` with a message naming
it:
```yaml
spec:
mode: Managed
provider: gcp-eu
placement:
zone: <ZONE> # e.g. europe-west1-b
machineType: e2-micro
image: projects/debian-cloud/global/images/family/debian-12
```
```sh
kubectl apply -f config/samples/proxy_gcp.yaml
```
## 6. What you should see
```sh
kubectl get proxy -w
```
`Provisioning` → `Running` (VM's external IP published in status) →
`Ready` (CONNECT health probe succeeded through the public IP). Then:
```sh
# the VM exists and carries the GC labels
gcloud compute instances list --project <PROJECT_ID> \
--filter 'labels.proxy-operator-managed=yes'
# the proxy actually tunnels — should print the VM's external IP
curl -x http://<EXTERNAL_IP>:3128 https://ifconfig.me
```
Cleanup — the finalizer deletes the VM:
```sh
kubectl delete proxy proxy-gcp-sample
gcloud compute instances list --project <PROJECT_ID> # should be empty again
```
## Gotchas
- **The orphan GC sweeps the whole project**: any VM labeled
`proxy-operator-managed=yes` whose UID does not match a live Proxy CR
in *this* cluster is deleted once past the age threshold. Do not point
two operator installs at the same project, and do not hand-create VMs
with that label.
- **VM creation is fire-and-forget** — the provider never waits on the
insert operation; progress is discovered by polling `Get`. A quota
error or bad image name surfaces on the Proxy's status/conditions a
reconcile later, not synchronously. `kubectl describe proxy` is the
place to look when something stalls.
- **e2-micro costs pennies but is not free everywhere** — remember to
delete the CR (or check `gcloud compute instances list`) when done.

View File

@@ -4,9 +4,10 @@ Plan: `docs/plans/2026-08-11-1838-gcp-http-wire-logging-v5.md`
- [x] Step 1 — `wireLogger` + `option.WithLogger` wiring in `internal/provider/gcp/gcp.go` - [x] Step 1 — `wireLogger` + `option.WithLogger` wiring in `internal/provider/gcp/gcp.go`
- [x] Step 2 — Tests (`TestWireLogger_gatesAtV5`, `TestWireLogger_infoLandsAtV1`) - [x] Step 2 — Tests (`TestWireLogger_gatesAtV5`, `TestWireLogger_infoLandsAtV1`)
- [ ] Step 3 — Live verification at `--zap-log-level=5` (user, on cluster) - [x] Step 3 — Live verification at `--zap-log-level=5` (user, on cluster)
- [ ] Step 4CHANGELOG entry (after live confirmation; batch with the two - [x] Step 3bPost-verification fix: drop auth records, elide huge fields
earlier pending entries: GCP V-logging, version stamp) - [ ] Step 4 — CHANGELOG entry (after live confirmation of 3b; batch with the
two earlier pending entries: GCP V-logging, version stamp)
## Steps 12 ## Steps 12
@@ -35,3 +36,28 @@ Verified with:
go test -race ./internal/provider/gcp/ go test -race ./internal/provider/gcp/
go build ./... && go test ./... go build ./... && go test ./...
``` ```
## Step 3b — what live verification exposed, and the fix
Live V(5) output revealed two problems the plan missed:
1. **Security: the injected logger propagates into `cloud.google.com/go/auth`**,
which logs its own token exchange (`auth.go:571/576`) — signed JWT
assertion in the request, full bearer access token in the response. The
plan's "auth token is safe" analysis only covered the compute client's
request headers, not the auth library's own records. Fix: `wireLogger`
now wraps the handler in a filter that drops every Debug record except
the compute client's `"api request"`/`"api response"` (allowlist, so
future SDK additions fail closed); Warn/Error still pass through.
2. **Readability: GCP responses embed multi-KB blobs** (Shielded-VM UEFI
dbx databases) that swamp the line. Fix: string fields >1KiB are elided
to `[elided N bytes]` by default, recursively through payload
maps/arrays. Opt-out via new manager flag
`--gcp-wire-log-full-payloads` (threaded through a constructor closure
in `cmd/main.go``gcp.NewWithWireOptions`; the `registry.Constructor`
signature stays unchanged). Chosen by the user: elision on by default,
verbatim available on demand. Auth records are dropped in both modes.
The filter/elision logic lives in `internal/provider/gcp/wirelog.go` with
tests covering: auth-record drop (both modes), elision marker + small-field
preservation, verbatim mode, and the original V(5) gating.

View File

@@ -0,0 +1,53 @@
# Execution: Distilled Gitea Actions image-build workflow
Plan: `docs/plans/2026-08-11-1935-gitea-build-workflow.md`
- [x] Step 1 — Create `.gitea/workflows/build.yaml`
- [x] Step 2 — Replace CLAUDE.md CI TODO with a CI/CD subsection
- [x] Step 3 — Push branch + open MR
- [ ] Step 4 — CHANGELOG entry (after the first successful run is confirmed)
## Steps 12 — workflow + CLAUDE.md
The workflow distills the house pattern from 9 sibling projects (survey in the plan)
plus improvements none of them combine: an immutable `sha-<12>` tag, a lightweight
test gate, `:latest` moving only on real tag pushes, and a `concurrency` group.
Work happened in a git worktree off `origin/main` so the main checkout (which had
unrelated uncommitted changes) stayed untouched:
```bash
git worktree add -b feat/gitea-build-workflow \
"$SCRATCH/wt-build" origin/main
```
Deviation from the plan's assumptions: while planning, `feat/proxy-operator` was
still unmerged and the plan noted `main` lacked `internal/`/`test/`. By execution
time `origin/main` had moved (`076bc66..f7000f7` — the proxy-operator MR merged),
so the branch and its CI gate cover the full operator code.
Verified the workflow parses and the check-gate commands pass on this exact tree
(ruby stands in for a YAML linter because the system python3 has no `yaml` module):
```bash
ruby -ryaml -e "YAML.load_file('.gitea/workflows/build.yaml'); puts 'YAML OK'"
go vet ./... && go build ./... && go test -short ./... # all packages ok
```
## Step 3 — push + MR
Branch pushed and MR opened with `tea` (the worktree was then removed and the main
checkout switched onto the branch so the files are visible locally):
```bash
tea pr create --title "Add Gitea Actions image-build workflow" \
--description "..." --base main --head feat/gitea-build-workflow
# → https://gitea.home.hrajfrisbee.cz/kacerr/egress-proxies-operator/pulls/2
```
Worth noting: the workflow itself cannot run end-to-end until (a) the MR merges
(it only triggers on tags / manual dispatch, not branch pushes) and (b) the
`REGISTRY_TOKEN` secret is created in this repo's Gitea settings (PAT with
`write:package`). First real verification = manual dispatch with tag
`manual-test`, expecting `manual-test` + `sha-…` in Packages and `:latest`
untouched.

View File

@@ -0,0 +1,135 @@
# Plan: Distilled Gitea Actions image-build workflow
**Created:** 2026-08-11 19:35
## Context
This repo (`egress-proxies-operator`) has a Dockerfile, a Makefile with `docker-build`/`docker-push` targets, and a Gitea remote — but no CI workflow (CLAUDE.md flags this as a TODO). A survey of all projects under `/Users/jan.novak/srv` found 9 image-build workflows, all variations of one lineage: trigger on `workflow_dispatch` + tag push, `docker login` to `gitea.home.hrajfrisbee.cz` with `secrets.REGISTRY_TOKEN`, raw `docker build`/`docker push`, `runs-on: ubuntu-latest`, `permissions: {contents: read, packages: write}`.
The best individual ideas are scattered:
- **aviso_v2**: quality-gate job before build; computes `sha-<short>` as a second immutable tag via `$GITHUB_OUTPUT`.
- **gateway-helper-operator** (closest sibling — same kubebuilder shape): passes build args (`GIT_COMMIT` etc.), tags `:latest` alongside the version tag.
- **psmf-data-sync test.yaml**: `actions/setup-go@v5` with `go-version-file: go.mod` + module cache (proven to work on the act_runner).
Goal: distill these into one `build.yaml` for this repo. User decisions: triggers = **tags + manual dispatch only** (house convention, no builds from main); build tool = **raw docker CLI** (the runner bind-mounts docker.sock, so this just works); **lightweight test gate** (`go vet` + `go build` + `go test -short`, no envtest download); **amd64 only**.
## The workflow
Create `.gitea/workflows/build.yaml`:
```yaml
name: Build and Push
on:
workflow_dispatch:
inputs:
tag:
description: 'Image tag'
required: true
default: 'latest'
push:
tags:
- '*'
concurrency:
group: build-${{ github.ref }}
cancel-in-progress: true
jobs:
check:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-go@v5
with:
go-version-file: go.mod
cache: true
- name: Vet
run: go vet ./...
- name: Build
run: go build ./...
- name: Test (short)
run: go test -short ./...
build:
needs: check
runs-on: ubuntu-latest
permissions:
contents: read
packages: write
steps:
- uses: actions/checkout@v4
- name: Compute image tags
id: meta
run: |
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
TAG="${{ inputs.tag }}"
else
TAG="${{ github.ref_name }}"
fi
echo "tag=$TAG" >> "$GITHUB_OUTPUT"
echo "sha=sha-$(echo '${{ github.sha }}' | cut -c1-12)" >> "$GITHUB_OUTPUT"
- name: Login to Gitea registry
run: echo "${{ secrets.REGISTRY_TOKEN }}" | docker login -u ${{ github.actor }} --password-stdin gitea.home.hrajfrisbee.cz
- name: Build and push
run: |
IMAGE=gitea.home.hrajfrisbee.cz/${{ github.repository }}
docker build \
--build-arg GIT_COMMIT=$(echo '${{ github.sha }}' | cut -c1-12) \
--label org.opencontainers.image.source=https://gitea.home.hrajfrisbee.cz/${{ github.repository }} \
--label org.opencontainers.image.created=$(date -u +%Y-%m-%dT%H:%M:%SZ) \
-t "$IMAGE:${{ steps.meta.outputs.tag }}" \
-t "$IMAGE:${{ steps.meta.outputs.sha }}" \
.
docker push "$IMAGE:${{ steps.meta.outputs.tag }}"
docker push "$IMAGE:${{ steps.meta.outputs.sha }}"
- name: Push latest (tag builds only)
if: github.event_name == 'push'
run: |
IMAGE=gitea.home.hrajfrisbee.cz/${{ github.repository }}
docker tag "$IMAGE:${{ steps.meta.outputs.tag }}" "$IMAGE:latest"
docker push "$IMAGE:latest"
```
### What's distilled vs. improved over the existing workflows
Distilled (house patterns kept as-is): triggers, `REGISTRY_TOKEN` + `github.actor` login, image name `gitea.home.hrajfrisbee.cz/${{ github.repository }}`, `ubuntu-latest`, raw docker CLI, `permissions` block.
Improvements none of the existing workflows have all of:
1. **`sha-<12>` immutable tag** alongside the human tag (aviso_v2 had this; nobody else) — lets deployments pin exactly what was built.
2. **Test gate** (aviso_v2 had one; the Go projects don't) — lightweight variant per user choice; uses `go-version-file: go.mod` so the Go version never drifts from the module.
3. **`GIT_COMMIT` build arg** matches the Makefile/Dockerfile contract — the binary's `internal/version.Commit` and the `org.opencontainers.image.revision` label get the real commit (12-char, same width as the Makefile's `git rev-parse --short=12`; no `-dirty` needed since CI checkouts are clean).
4. **`:latest` only on real tag pushes**, not manual dispatch — gateway-helper pushed `latest` unconditionally, which lets an ad-hoc dispatch of an old ref clobber `latest`.
5. **`concurrency` group** — cancels a superseded run of the same ref (none of the 20 surveyed workflows have this).
6. **OCI `source`/`created` labels** added at build time (revision label already comes from the Dockerfile).
## Files
- **Create** `.gitea/workflows/build.yaml` — content above.
- **Update** `CLAUDE.md` — replace the `TODO: no .gitea/workflows/ CI pipeline exists yet` note in the Git Commits section with a short CI/CD subsection describing the workflow (triggers, secret, tags produced).
- **Update** `CHANGELOG.md` — new top entry (after user confirms it works, per convention; timestamp via `date "+%Y-%m-%d %H:%M %Z"`).
- Copy this plan to `docs/plans/YYYY-MM-DD-HHMM-gitea-build-workflow.md` (timestamp via `date "+%Y-%m-%d-%H%M"`) and commit it first, per CLAUDE.md ordering rule.
## Branch & MR
House convention: feature → own branch + MR. Dockerfile and `cmd/` already exist on `main`, so:
1. `git checkout -b feat/gitea-build-workflow origin/main` (do not touch the current `feat/proxy-operator` branch's uncommitted `.claude/settings.json` change — leave it be).
2. Commit plan file, then the workflow + CLAUDE.md update (with `Co-Authored-By: Claude <noreply@anthropic.com>`).
3. `git push -u origin feat/gitea-build-workflow`, open MR with `tea pr create --base main --head feat/gitea-build-workflow`. Do not merge.
Note: `main` has no `internal/`/`test/` dirs yet (those are on `feat/proxy-operator`), which is fine — the workflow only fires on tags/dispatch, and by then the operator branch will be merged. `go build ./...` / `go test -short ./...` work on both branch states.
## Prerequisite (user action)
`REGISTRY_TOKEN` secret must exist in this repo's Gitea settings (Settings → Actions → Secrets): a personal access token with `write:package` scope — same as every other project uses. Flag this in the MR description.
## Verification
The workflow doesn't trigger on branch pushes, so end-to-end verification happens after merge:
1. Local sanity: `docker build --build-arg GIT_COMMIT=test -t scratch-check .` (confirms the build args/labels line is valid) — or at minimum a YAML parse check.
2. After the MR merges: run the workflow manually via Gitea UI (Actions → Build and Push → Run workflow, tag `manual-test`), confirm both `manual-test` and `sha-…` tags appear under Packages, and that `:latest` was NOT updated.
3. Then push a real version tag (e.g. `v0.1.0`) and confirm `v0.1.0`, `sha-…`, and `latest` all appear.

View File

@@ -0,0 +1,131 @@
# PR review findings: feat/proxy-operator
**Created:** 2026-08-10 11:34
**Scope:** `origin/main...feat/proxy-operator` (merge-base 076bc66, 25 commits, ~80 files)
**Reviewers:** `go-operator-reviewer` + `operator-reviewer` agents; findings consolidated, most severe first. Check off items as they're processed.
Both reviewers rated the core reconcile architecture sound: single status writer with one
deferred patch, finalizer added before any provider call, Get-before-RemoveFinalizer on
delete, CEL immutability rules correctly split to avoid the oldSelf-on-CREATE trap,
leader-election gating on destructive runnables, GC tombstone rules (MinAge, UID-less
instances never deleted).
## Merge-blockers
- [ ] **Discovery leases proxies with an empty IP** — found independently by both reviewers.
`internal/discovery/handlers.go:151`, `internal/controller/proxy_controller.go:226`
During instance replacement (and the Get→NotFound recovery path) the reconciler clears
`status.ip` but only the create branch removes the `Healthy` condition, and the health
engine prunes state for empty-host proxies so nothing refreshes it. For the whole
delete→recreate window (minutes on GCP), `isHealthy` still returns true and
`handleAcquireLease` grants `201 Created` with `"ip": ""`, burning a `MaxLeases` slot.
**Fix:** add `EffectiveHost() != ""` to `isHealthy` (covers list + acquire), and
remove/downgrade `Healthy` wherever `status.IP` is cleared.
- [ ] **Orphan GC deletes other installations' fleets in a shared GCP project.**
`internal/provider/gcp/insert.go:57`, `internal/gc/gc.go:93`
Instances are tagged only `proxy-operator-managed=true` + CR UID; the sweeper deletes any
tagged instance whose UID isn't in *its own cluster's* Proxy list. Two clusters sharing a
GCP project delete each other's VMs every GC interval in a permanent loop.
**Fix:** add an installation-identity label (cluster/deployment ID) set by both providers
and filtered on in `ListByTag`.
- [ ] **Permanent-error latch wedges proxies on failures that aren't spec-caused.**
`internal/controller/proxy_controller.go:126`
Latch keys on `observedGeneration == generation`, but two failure inputs live outside the
spec: an unconfigured provider (config fix + restart doesn't bump generation, and
`spec.provider` is CEL-immutable → stuck `Failed` short of deleting the CR) and resolved
Secret content (Secret fix enqueues a reconcile that short-circuits at the latch before
re-resolving cloud-init).
**Fix:** latch should also consider current spec-hash / provider availability.
## Worth fixing
- [ ] **Deletion-path failures invisible in status** — flagged by both reviewers.
`internal/controller/proxy_controller.go:319`, `:266`
`deletionFailure` swallows `ErrQuotaExceeded` (nil error, no status write); unconfigured
provider returns a bare error forever. A Proxy wedged in `Deleting` shows nothing in
`kubectl describe`. Stage `setProvisioned(p, False, ReasonDeleting, ...)` before returning.
Also: `Delete` is resubmitted on every `DeletionPoll` pass, churning GCP quota — a state
check on the `Get` result would avoid it.
- [ ] **Lost providerID on `setSpecHash` conflict.**
`internal/controller/proxy_controller.go:161`
On Update conflict the function returns before `p.Status.ProviderID = id`, so the deferred
patch persists an empty providerID for a just-created instance. Self-heals via GC.
**Fix:** set `p.Status.ProviderID = id` before returning the error (one line).
- [ ] **Terminating pods still report `StateRunning`.**
`internal/provider/kubernetes/kubernetes.go:151`
A pod with a deletionTimestamp keeps `phase=Running` + `PodIP` while terminating, so drift
reconcile republishes `Provisioned=True` and discovery keeps leasing a dying pod.
**Fix:** map non-zero `pod.DeletionTimestamp` to `StateTerminated` in `instanceFromPod`.
- [ ] **Stale-cache spec-hash race deletes the freshly created replacement instance.**
`internal/controller/proxy_controller.go:176`
Instance name derives from CR UID, so old and new instances share a providerID. A reconcile
served a cached object from before a just-completed replacement re-enters `replaceInstance`
and deletes the *new* healthy instance. Converges, but destroys a good instance.
**Fix:** re-read uncached before the destructive branch, or compare `inst.CreatedAt`
against the annotation-update time.
- [ ] **`observedGeneration` written before the generation is actually processed.**
`internal/controller/status.go:114`
Set unconditionally in `patchStatusIfChanged`, including on the finalizer-add pass and
`resolveCloudInit` failures — misleads kstatus-style tooling. Set it only once the state
machine has genuinely evaluated the spec.
- [ ] **No event filtering on the Proxy watch.**
`internal/controller/proxy_controller.go:402`
Every self-inflicted status patch triggers a follow-up reconcile with an extra cloud `Get`,
roughly doubling provider read traffic. Caution: a plain `GenerationChangedPredicate`
breaks the finalizer flow (relies on its own Update event to re-enter) — needs a
status-only/resourceVersion-only filter or an explicit requeue in the finalizer pass.
- [ ] **Unlabelled cloud-init Secrets produce a misleading NotFound with endless backoff.**
`internal/controller/proxy_controller.go:344`, `cmd/main.go:210`
The label-restricted cache turns "exists but missing `crawl.example.com/cloud-init=true`"
into `CloudInitError: not found`. Mention the label requirement in the condition message,
or read via uncached `APIReader` and validate the label explicitly.
- [ ] **RBAC over-grant.**
`config/rbac/role.yaml:25`
`create;delete` on `proxies` is scaffold residue (controller never creates/deletes CRs);
cluster-wide `pods create/delete` and `secrets get/list/watch` apply even when only the GCP
provider is configured — pod rules belong in an optional kustomize component.
## Simplifications
- [ ] **Delete `internal/provider/registry`** — 14 lines of logic, one caller
(`cmd/main.go:148`); fold `Build`/`Constructor` into the composition root. Also fixes the
two-sources-of-truth problem: `internal/provider/config.go:84` hardcodes
`"kubernetes"`/`"gcp"` while `registry.Build` dispatches through a caller-supplied map —
validate against the constructor map instead. Net 1 package, 44 lines, 92 test lines.
- [ ] **Collapse `LeaseStore` interface to `*lease.Store`.**
`internal/discovery/server.go:28`
Single implementation and not a test seam (tests wire the real `lease.NewStore`).
Keep `HealthSnapshotter` and `instancesAPI` — those are genuine seams.
- [ ] **Replace metrics nil-guards with no-op defaults.**
`internal/health/engine.go:68`, `internal/discovery/server.go:38`,
`internal/provider/metrics.go:8`
Keep the interfaces (legit "no prometheus in domain packages" rationale) but default the
fields to a no-op impl — `provider.WithMetrics` already dereferences unconditionally, so
the guards are inconsistent anyway.
## Nice-to-have
- [ ] Add an `OwnerReference` to provider pods (`internal/provider/kubernetes/pod.go:24`) —
free cascading deletion if the finalizer is ever bypassed; `CreateRequest` already carries
Namespace/ProxyName/UID.
- [ ] `Close()` the GCP `*compute.InstancesClient` (`internal/provider/gcp/gcp.go:89`) —
harmless today, a leak the moment providers are rebuilt on config reload.
- [ ] Fix `.golangci` config: it references a missing `logcheck` plugin, so the linter only
runs with the project config disabled.
- [ ] External-mode endpoint edits don't reset health-engine counters
(`internal/health/engine.go:206` keys on name+UID): flipping `endpoint.host` keeps the old
host's `Healthy=True` for `failureThreshold × interval`. Arguably a replacement, not a flap.
- [ ] `init()` funcs at `api/v1alpha1/proxy_types.go:353` and `cmd/main.go:67` conflict with
the repo's "no `init()`" convention; kubebuilder-idiomatic, but scheme registration could
use the scaffold's `SchemeBuilder.Register` at package var scope.

View File

@@ -9,7 +9,6 @@ package gcp
import ( import (
"context" "context"
"fmt" "fmt"
"log/slog"
"strings" "strings"
"time" "time"
@@ -85,28 +84,27 @@ type Provider struct {
api instancesAPI api instancesAPI
} }
// wireLogger returns the slog logger handed to the SDK: its Debug-level
// "api request"/"api response" records (slog Debug = +4 on the logr
// scale) land at V(5) on top of the base's V(1) shift.
func wireLogger(base logr.Logger) *slog.Logger {
return slog.New(logr.ToSlogHandler(base.V(1)))
}
// New builds a Provider using Application Default Credentials (workload // New builds a Provider using Application Default Credentials (workload
// identity in-cluster, gcloud ADC locally — no key-file plumbing). // identity in-cluster, gcloud ADC locally — no key-file plumbing).
// Deliberately untested: it dials real Google endpoints; everything below // Deliberately untested: it dials real Google endpoints; everything below
// it is exercised through newWithAPI. // it is exercised through newWithAPI.
//
// The injected wire logger surfaces the SDK's raw HTTP request/response
// records at V(5); note option.WithLogger overrides the SDK's own
// GOOGLE_SDK_GO_LOGGING_LEVEL env var, so --zap-log-level is the only knob.
func New(ctx context.Context, pc provider.ProviderConfig) (provider.Provider, error) { func New(ctx context.Context, pc provider.ProviderConfig) (provider.Provider, error) {
return NewWithWireOptions(ctx, pc, WireLogOptions{})
}
// NewWithWireOptions is New with explicit control over the V(5) wire
// logging; the injected wire logger surfaces the SDK's HTTP
// request/response records at V(5). Note option.WithLogger overrides the
// SDK's own GOOGLE_SDK_GO_LOGGING_LEVEL env var, so --zap-log-level is
// the only knob.
func NewWithWireOptions(ctx context.Context, pc provider.ProviderConfig, opts WireLogOptions) (provider.Provider, error) {
base := logf.Log.WithName("gcp").WithName("http") base := logf.Log.WithName("gcp").WithName("http")
if base.V(5).Enabled() { if base.V(5).Enabled() {
logf.Log.WithName("gcp").Info( logf.Log.WithName("gcp").Info(
"GCP HTTP wire logging active — request payloads include cloud-init user-data") "GCP HTTP wire logging active — request payloads include cloud-init user-data",
"fullPayloads", opts.FullPayloads)
} }
client, err := compute.NewInstancesRESTClient(ctx, option.WithLogger(wireLogger(base))) client, err := compute.NewInstancesRESTClient(ctx, option.WithLogger(wireLogger(base, opts)))
if err != nil { if err != nil {
return nil, fmt.Errorf("creating GCP instances client: %w", err) return nil, fmt.Errorf("creating GCP instances client: %w", err)
} }

View File

@@ -3,6 +3,7 @@ package gcp
import ( import (
"context" "context"
"errors" "errors"
"log/slog"
"strings" "strings"
"testing" "testing"
"time" "time"
@@ -418,7 +419,7 @@ func TestWireLogger_gatesAtV5(t *testing.T) {
*lines = append(*lines, prefix+" "+args) *lines = append(*lines, prefix+" "+args)
}, funcr.Options{Verbosity: tc.verbosity}) }, funcr.Options{Verbosity: tc.verbosity})
slogger := wireLogger(base) slogger := wireLogger(base, WireLogOptions{})
slogger.Debug("api request", "rpcName", "Insert") slogger.Debug("api request", "rpcName", "Insert")
joined := strings.Join(*lines, "\n") joined := strings.Join(*lines, "\n")
@@ -439,13 +440,84 @@ func TestWireLogger_infoLandsAtV1(t *testing.T) {
*lines = append(*lines, prefix+" "+args) *lines = append(*lines, prefix+" "+args)
}, funcr.Options{Verbosity: 1}) }, funcr.Options{Verbosity: 1})
wireLogger(base).Info("hello") wireLogger(base, WireLogOptions{}).Info("hello")
if joined := strings.Join(*lines, "\n"); !strings.Contains(joined, "hello") { if joined := strings.Join(*lines, "\n"); !strings.Contains(joined, "hello") {
t.Errorf("slog Info should land at V(1) and be visible at verbosity 1; output:\n%s", joined) t.Errorf("slog Info should land at V(1) and be visible at verbosity 1; output:\n%s", joined)
} }
} }
func captureWireLogger(verbosity int, opts WireLogOptions) (*slog.Logger, *[]string) {
lines := &[]string{}
base := funcr.New(func(prefix, args string) {
*lines = append(*lines, prefix+" "+args)
}, funcr.Options{Verbosity: verbosity})
return wireLogger(base, opts), lines
}
func TestWireLogger_dropsNonAPIDebugRecords(t *testing.T) {
t.Parallel()
slogger, lines := captureWireLogger(9, WireLogOptions{})
const secret = "assertion=eyJhbGciOiJSUzI1NiJ9.SECRET"
slogger.Debug("2LO token request", "request", map[string]any{"payload": secret})
slogger.Debug("2LO token response", "response", map[string]any{"payload": "ya29.SECRET-TOKEN"})
if len(*lines) != 0 {
t.Errorf("auth token-exchange records must be dropped; got:\n%s", strings.Join(*lines, "\n"))
}
slogger.Warn("credential refresh failed")
if joined := strings.Join(*lines, "\n"); !strings.Contains(joined, "credential refresh failed") {
t.Errorf("non-debug SDK records should pass through; output:\n%s", joined)
}
}
func TestWireLogger_elidesLargeFields(t *testing.T) {
t.Parallel()
slogger, lines := captureWireLogger(9, WireLogOptions{})
huge := strings.Repeat("x", 4096)
slogger.Debug("api response", "response", map[string]any{
"status": "200",
"payload": map[string]any{
"name": "proxy-abc",
"disks": []any{map[string]any{"content": huge}},
},
})
joined := strings.Join(*lines, "\n")
if strings.Contains(joined, huge[:64]) {
t.Errorf("large field not elided:\n%.500s", joined)
}
if !strings.Contains(joined, "[elided 4096 bytes]") {
t.Errorf("elision marker missing:\n%s", joined)
}
for _, keep := range []string{"proxy-abc", "200", "api response"} {
if !strings.Contains(joined, keep) {
t.Errorf("small field %q lost during elision:\n%s", keep, joined)
}
}
}
func TestWireLogger_fullPayloadsDisablesElision(t *testing.T) {
t.Parallel()
slogger, lines := captureWireLogger(9, WireLogOptions{FullPayloads: true})
huge := strings.Repeat("y", 4096)
slogger.Debug("api response", "response", map[string]any{"payload": huge})
joined := strings.Join(*lines, "\n")
if !strings.Contains(joined, huge) {
t.Errorf("FullPayloads should keep fields verbatim:\n%.200s", joined)
}
slogger.Debug("2LO token response", "response", "ya29.SECRET")
if joined := strings.Join(*lines, "\n"); strings.Contains(joined, "ya29.SECRET") {
t.Error("auth records must be dropped even with FullPayloads")
}
}
func TestLogging_apiErrorKeepsHTTPDetail(t *testing.T) { func TestLogging_apiErrorKeepsHTTPDetail(t *testing.T) {
t.Parallel() t.Parallel()
ctx, lines := captureContext(1) ctx, lines := captureContext(1)

View File

@@ -0,0 +1,117 @@
package gcp
import (
"context"
"fmt"
"log/slog"
"github.com/go-logr/logr"
)
// wireLogMaxFieldBytes is the elision threshold for string fields in wire
// payloads: GCP responses embed multi-KB blobs (Shielded-VM UEFI dbx
// databases, licenses) that swamp the log line without diagnostic value.
const wireLogMaxFieldBytes = 1024
// WireLogOptions controls the V(5) HTTP wire logging of the GCP SDK.
type WireLogOptions struct {
// FullPayloads disables field elision and logs payloads verbatim.
FullPayloads bool
}
// wireLogger returns the slog logger handed to the SDK: its Debug-level
// "api request"/"api response" records (slog Debug = +4 on the logr
// scale) land at V(5) on top of the base's V(1) shift.
//
// Debug records other than the compute client's api request/response are
// dropped entirely: the same logger propagates into the auth library,
// whose token-exchange records contain the signed JWT assertion and the
// bearer access token. Warnings and errors pass through.
func wireLogger(base logr.Logger, opts WireLogOptions) *slog.Logger {
return slog.New(&wireFilterHandler{
inner: logr.ToSlogHandler(base.V(1)),
fullPayloads: opts.FullPayloads,
})
}
type wireFilterHandler struct {
inner slog.Handler
fullPayloads bool
}
func (h *wireFilterHandler) Enabled(ctx context.Context, level slog.Level) bool {
return h.inner.Enabled(ctx, level)
}
func (h *wireFilterHandler) Handle(ctx context.Context, rec slog.Record) error {
if rec.Level <= slog.LevelDebug && rec.Message != "api request" && rec.Message != "api response" {
return nil
}
if h.fullPayloads {
return h.inner.Handle(ctx, rec)
}
elided := slog.NewRecord(rec.Time, rec.Level, rec.Message, rec.PC)
rec.Attrs(func(a slog.Attr) bool {
elided.AddAttrs(slog.Attr{Key: a.Key, Value: elideValue(a.Value)})
return true
})
return h.inner.Handle(ctx, elided)
}
func (h *wireFilterHandler) WithAttrs(attrs []slog.Attr) slog.Handler {
return &wireFilterHandler{inner: h.inner.WithAttrs(attrs), fullPayloads: h.fullPayloads}
}
func (h *wireFilterHandler) WithGroup(name string) slog.Handler {
return &wireFilterHandler{inner: h.inner.WithGroup(name), fullPayloads: h.fullPayloads}
}
func elideValue(v slog.Value) slog.Value {
v = v.Resolve()
switch v.Kind() {
case slog.KindString:
if s := v.String(); len(s) > wireLogMaxFieldBytes {
return slog.StringValue(elisionMarker(len(s)))
}
return v
case slog.KindGroup:
attrs := v.Group()
out := make([]slog.Attr, 0, len(attrs))
for _, a := range attrs {
out = append(out, slog.Attr{Key: a.Key, Value: elideValue(a.Value)})
}
return slog.GroupValue(out...)
case slog.KindAny:
return slog.AnyValue(elideAny(v.Any()))
default:
return v
}
}
func elideAny(v any) any {
switch t := v.(type) {
case string:
if len(t) > wireLogMaxFieldBytes {
return elisionMarker(len(t))
}
return t
case map[string]any:
out := make(map[string]any, len(t))
for k, val := range t {
out[k] = elideAny(val)
}
return out
case []any:
out := make([]any, len(t))
for i, val := range t {
out[i] = elideAny(val)
}
return out
default:
return v
}
}
func elisionMarker(size int) string {
return fmt.Sprintf("[elided %d bytes]", size)
}