mirror of
https://hubproxy.babadafafafafa.cn/https://github.com/kube-vip/kube-vip.git
synced 2026-09-20 08:03:47 +08:00
Compare commits
149 Commits
SNAT_fixes
...
b514ae2733
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b514ae2733 | ||
|
|
26eab74f3e | ||
|
|
44a67bc901 | ||
|
|
7df2e5dd46 | ||
|
|
ba8ffde3c8 | ||
|
|
65a6a6f8f2 | ||
|
|
94ff8495f4 | ||
|
|
52021d9232 | ||
|
|
776f0b18fa | ||
|
|
86ba1d09af | ||
|
|
6512709032 | ||
|
|
1196748efc | ||
|
|
d557211b55 | ||
|
|
e4eb5e8e7f | ||
|
|
a7b068f9d2 | ||
|
|
a6bf5280d4 | ||
|
|
61a52e8f2a | ||
|
|
7e2dd0d262 | ||
|
|
7a085cb3ce | ||
|
|
1f5c135fd1 | ||
|
|
b4771c5319 | ||
|
|
3771ccee29 | ||
|
|
ee64dceb36 | ||
|
|
38fabeba9e | ||
|
|
7ef1899567 | ||
|
|
637c3da47e | ||
|
|
e666a0cdd1 | ||
|
|
9a2142c028 | ||
|
|
0606e9477c | ||
|
|
8a968618bf | ||
|
|
5cd138a85b | ||
|
|
246a786fe2 | ||
|
|
6ee3024bc6 | ||
|
|
2bc2df53fc | ||
|
|
618904dea3 | ||
|
|
b116d5a469 | ||
|
|
6cbf5aaeda | ||
|
|
dc453f07fc | ||
|
|
14b2f51aba | ||
|
|
01e8fbc3e3 | ||
|
|
4504649c91 | ||
|
|
47f4e75183 | ||
|
|
f589a18bd9 | ||
|
|
d79f3ddb52 | ||
|
|
1b25ce7d0e | ||
|
|
96c4406d63 | ||
|
|
a51944b89d | ||
|
|
a36dc36947 | ||
|
|
47b546073a | ||
|
|
2b126dceed | ||
|
|
110d34b844 | ||
|
|
5388fed92b | ||
|
|
487859e76f | ||
|
|
0e57deef3d | ||
|
|
b25badd185 | ||
|
|
77fa726c99 | ||
|
|
920a0182cb | ||
|
|
8d043de910 | ||
|
|
b0b12cfc13 | ||
|
|
b034c81bef | ||
|
|
ca9640227a | ||
|
|
5276f0123f | ||
|
|
c2e5be6d6a | ||
|
|
a8293f66e4 | ||
|
|
9d72f43f62 | ||
|
|
3c3096d89d | ||
|
|
5d5c893501 | ||
|
|
d417a0c8e8 | ||
|
|
825ffdb20c | ||
|
|
fe3a379f2d | ||
|
|
c42a207c26 | ||
|
|
d6e5753464 | ||
|
|
5131b92810 | ||
|
|
e1fd9ac3e3 | ||
|
|
1831a05525 | ||
|
|
19198d47fa | ||
|
|
1af388ff9f | ||
|
|
d1ff5f2952 | ||
|
|
54881a117c | ||
|
|
426a409a5d | ||
|
|
5871bec56c | ||
|
|
b7f3379514 | ||
|
|
a15745c442 | ||
|
|
b684eed5a4 | ||
|
|
0bdd6a9015 | ||
|
|
b864abf27d | ||
|
|
6fa38027e2 | ||
|
|
b478234d29 | ||
|
|
03241ee27f | ||
|
|
6e8b391685 | ||
|
|
e5ff483a23 | ||
|
|
bee5cfe4a2 | ||
|
|
81d54050c7 | ||
|
|
5e9fcf642c | ||
|
|
83797e4da1 | ||
|
|
ed28c49f48 | ||
|
|
42ba1fefb2 | ||
|
|
988eb0994a | ||
|
|
90a3892271 | ||
|
|
c553663654 | ||
|
|
b6151a4454 | ||
|
|
6f69d4511f | ||
|
|
5e2220fd4d | ||
|
|
4e13a81af0 | ||
|
|
a19500b116 | ||
|
|
85a8c94ac5 | ||
|
|
150983ddd0 | ||
|
|
7911dcf3b9 | ||
|
|
9748f6366c | ||
|
|
ea916c5a31 | ||
|
|
6376d89fea | ||
|
|
5b2a62a10b | ||
|
|
202589cd33 | ||
|
|
0040633d89 | ||
|
|
dd022d89bb | ||
|
|
5b01e0aba7 | ||
|
|
7ce55caffa | ||
|
|
85f1c90bcf | ||
|
|
60cea74703 | ||
|
|
530c602152 | ||
|
|
4577f5bbe2 | ||
|
|
2572482658 | ||
|
|
15a8ca3881 | ||
|
|
7f0069a58c | ||
|
|
cfd86de936 | ||
|
|
9ff88eba50 | ||
|
|
fdbad81da2 | ||
|
|
d1fa3a20ec | ||
|
|
1e81d048b7 | ||
|
|
bedbba70a7 | ||
|
|
4f32829ab0 | ||
|
|
649f5f08e8 | ||
|
|
49d815775f | ||
|
|
68123d30dc | ||
|
|
c3ba4a8b64 | ||
|
|
39300b8513 | ||
|
|
d90db3ed5b | ||
|
|
972e0fd611 | ||
|
|
02260149f1 | ||
|
|
793766265b | ||
|
|
00e0282719 | ||
|
|
6c94ecce64 | ||
|
|
4f8f43a412 | ||
|
|
df84b047b9 | ||
|
|
0d248ba40f | ||
|
|
bf29c32e56 | ||
|
|
8c8490746a | ||
|
|
fd6006bb8b | ||
|
|
899a3e5fe8 |
13
.github/suggestion-comment.md
vendored
Normal file
13
.github/suggestion-comment.md
vendored
Normal file
@@ -0,0 +1,13 @@
|
||||
I'll help you add a suggestion. Unfortunately, I can't directly add a suggestion to an existing comment through the API. However, here's what I recommend:
|
||||
|
||||
**Option 1: Reply with a suggestion**
|
||||
Create a new comment with a suggested fix:
|
||||
|
||||
```suggestion
|
||||
failed to get an IPv6 address after %d attempt(s), giving up, error: %s
|
||||
```
|
||||
|
||||
**Option 2: Edit your existing comment**
|
||||
Update your comment to include the suggestion details pointing out that line 284 in the error message says "IPv4" but should say "IPv6" since this is the DHCPv6Client.
|
||||
|
||||
Would you like me to create a new reply comment with the suggestion instead?
|
||||
2
.github/workflows/anchore-syft.yml
vendored
2
.github/workflows/anchore-syft.yml
vendored
@@ -26,6 +26,6 @@ jobs:
|
||||
with:
|
||||
ref: ${{ github.ref_name }}
|
||||
- name: Anchore SBOM Action
|
||||
uses: anchore/sbom-action@v0.24.0
|
||||
uses: anchore/sbom-action@v0.24.2
|
||||
with:
|
||||
format: cyclonedx-json
|
||||
|
||||
45
.github/workflows/ci-pull-request.yaml
vendored
45
.github/workflows/ci-pull-request.yaml
vendored
@@ -1,45 +1,61 @@
|
||||
name: For each PR
|
||||
on:
|
||||
pull_request:
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
e2e-tests:
|
||||
runs-on: ubuntu-latest
|
||||
name: E2E tests
|
||||
timeout-minutes: 120
|
||||
env:
|
||||
GINKGO_PROCS: ${{ matrix.ginkgo-procs }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 3
|
||||
matrix:
|
||||
mode: ["arp", "rt", "bgp"]
|
||||
fail-fast: true
|
||||
include:
|
||||
- mode: arp
|
||||
ginkgo-procs: 4
|
||||
- mode: rt
|
||||
ginkgo-procs: 4
|
||||
- mode: bgp
|
||||
ginkgo-procs: 4
|
||||
steps:
|
||||
- name: Get current date
|
||||
id: date
|
||||
run: echo "::set-output name=date::$(date +'%Y-%m-%d-%H-%M')"
|
||||
run: echo "date=$(date +'%Y-%m-%d-%H-%M')" >> "$GITHUB_OUTPUT"
|
||||
- name: Ensure fs wont cause issues
|
||||
run: sudo sysctl fs.inotify.max_user_instances=8192 && sudo sysctl fs.inotify.max_user_watches=524288
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
- name: Install Go
|
||||
uses: actions/setup-go@v6
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: Build image locally
|
||||
run: make dockerx86Local
|
||||
- name: Run Manifest generation tests
|
||||
run: make manifest-test
|
||||
if: matrix.mode == 'arp'
|
||||
- name: Run ARP mode tests
|
||||
run: DOCKER_API_VERSION=1.48 E2E_KEEP_LOGS=true make e2e-tests-arp
|
||||
if: matrix.mode== 'arp'
|
||||
run: DOCKER_API_VERSION=1.48 E2E_KEEP_LOGS=true GINKGO_ARGS="--output-dir=/tmp --json-report=kube-vip-test-report-arp.json" make e2e-tests-arp
|
||||
if: matrix.mode == 'arp'
|
||||
- name: Run RT mode tests
|
||||
run: DOCKER_API_VERSION=1.48 E2E_KEEP_LOGS=true make e2e-tests-rt
|
||||
if: matrix.mode== 'rt'
|
||||
run: DOCKER_API_VERSION=1.48 E2E_KEEP_LOGS=true GINKGO_ARGS="--output-dir=/tmp --json-report=kube-vip-test-report-rt.json" make e2e-tests-rt
|
||||
if: matrix.mode == 'rt'
|
||||
- name: Get GoBGP binaries
|
||||
run: make get-gobgp
|
||||
if: matrix.mode== 'bgp'
|
||||
if: matrix.mode == 'bgp'
|
||||
- name: Run BGP mode tests
|
||||
run: sudo -E PATH=$PATH DOCKER_API_VERSION=1.48 E2E_KEEP_LOGS=true make e2e-tests-bgp
|
||||
if: matrix.mode== 'bgp'
|
||||
run: sudo -E PATH=$PATH DOCKER_API_VERSION=1.48 E2E_KEEP_LOGS=true GINKGO_ARGS="--output-dir=/tmp --json-report=kube-vip-test-report-bgp.json" make e2e-tests-bgp
|
||||
if: matrix.mode == 'bgp'
|
||||
- name: Change log directory permissions
|
||||
run: sudo chmod -R 755 /tmp/kube-vip-test*
|
||||
if: matrix.mode== 'bgp' && always()
|
||||
if: matrix.mode == 'bgp' && always()
|
||||
- name: Save logs
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
@@ -49,14 +65,15 @@ jobs:
|
||||
service-e2e-tests:
|
||||
runs-on: ubuntu-latest
|
||||
name: E2E service tests
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: Get current date
|
||||
id: date
|
||||
run: echo "::set-output name=date::$(date +'%Y-%m-%d-%H-%M')"
|
||||
run: echo "date=$(date +'%Y-%m-%d-%H-%M')" >> "$GITHUB_OUTPUT"
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
- name: Install Go
|
||||
uses: actions/setup-go@v6
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: Build image with iptables
|
||||
|
||||
31
.github/workflows/ci.yaml
vendored
31
.github/workflows/ci.yaml
vendored
@@ -1,11 +1,19 @@
|
||||
name: For each commit
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
tags: ['v*']
|
||||
pull_request:
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
jobs:
|
||||
validation:
|
||||
runs-on: ubuntu-latest
|
||||
name: Checks and linters
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Init
|
||||
run: sudo apt-get update && sudo apt-get install -y build-essential && sudo sysctl fs.inotify.max_user_instances=8192 && sudo sysctl fs.inotify.max_user_watches=524288
|
||||
@@ -17,31 +25,45 @@ jobs:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
- name: Install Go
|
||||
uses: actions/setup-go@v6
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: Verify gofmt
|
||||
run: |
|
||||
unformatted=$(gofmt -l .)
|
||||
if [ -n "$unformatted" ]; then
|
||||
echo "The following files are not gofmt-formatted:"
|
||||
echo "$unformatted"
|
||||
exit 1
|
||||
fi
|
||||
- name: All checks
|
||||
run: make check
|
||||
unit-tests:
|
||||
runs-on: ubuntu-latest
|
||||
name: Unit tests
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
- name: Install Go
|
||||
uses: actions/setup-go@v6
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: Run tests
|
||||
run: make unit-tests
|
||||
- name: Run privileged network tests
|
||||
run: |
|
||||
sudo -E env PATH="$PATH" KUBE_VIP_REQUIRE_NETNS=1 go test -race ./pkg/services ./pkg/vip ./pkg/instance \
|
||||
-run 'TestRecover|TestServiceAddressRetained|TestRetainControlPlaneVIPs|TestAddressProtocol|TestKubeVIPAddressProtocol|TestCleanupKubeVIPAddresses|TestMonitorDefaultInterfaceReturnsErrorWhenTestLinkIsSetDown|TestCleanupLinkAttachmentsOnlyDeletesOwnedVLAN'
|
||||
integration-tests:
|
||||
name: Integration tests
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
- name: Install Go
|
||||
uses: actions/setup-go@v6
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: Run tests
|
||||
@@ -49,10 +71,11 @@ jobs:
|
||||
image-vul-check:
|
||||
runs-on: ubuntu-latest
|
||||
name: Image vulnerability scan
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
- name: Build image with iptables
|
||||
- name: Build image
|
||||
run: make dockerx86Action
|
||||
- name: Run Trivy vulnerability scanner
|
||||
uses: aquasecurity/trivy-action@master
|
||||
|
||||
2
.github/workflows/codeql-analysis.yml
vendored
2
.github/workflows/codeql-analysis.yml
vendored
@@ -41,7 +41,7 @@ jobs:
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Install Go
|
||||
uses: actions/setup-go@v6
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
|
||||
|
||||
111
.github/workflows/nightly-e2e.yaml
vendored
Normal file
111
.github/workflows/nightly-e2e.yaml
vendored
Normal file
@@ -0,0 +1,111 @@
|
||||
name: Nightly e2e
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: '30 2 * * *'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
unit-coverage:
|
||||
runs-on: ubuntu-latest
|
||||
name: Unit tests with coverage
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
- name: Install Go
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: Run tests
|
||||
run: make unit-tests
|
||||
- name: Summarize coverage
|
||||
if: always()
|
||||
run: |
|
||||
if test -f coverage.out; then
|
||||
echo "### Unit coverage" >> "$GITHUB_STEP_SUMMARY"
|
||||
go tool cover -func=coverage.out | tail -1 >> "$GITHUB_STEP_SUMMARY"
|
||||
else
|
||||
echo "### Unit coverage: report missing" >> "$GITHUB_STEP_SUMMARY"
|
||||
fi
|
||||
- name: Upload coverage
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: unit-coverage
|
||||
path: coverage.out
|
||||
if-no-files-found: error
|
||||
if: always()
|
||||
etcd-e2e:
|
||||
runs-on: ubuntu-latest
|
||||
name: Etcd E2E tests
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: Ensure fs wont cause issues
|
||||
run: sudo sysctl fs.inotify.max_user_instances=8192 && sudo sysctl fs.inotify.max_user_watches=524288
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
- name: Install Go
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: go.mod
|
||||
- name: Build image locally
|
||||
run: make dockerx86Local
|
||||
- name: Prepare Etcd artifacts
|
||||
if: always()
|
||||
run: |
|
||||
mkdir -p /tmp/kube-vip-etcd-artifacts
|
||||
: > /tmp/kube-vip-etcd-artifacts/suite.log
|
||||
printf '[]\n' > /tmp/kube-vip-etcd-artifacts/report.json
|
||||
- name: Run Etcd tests
|
||||
id: etcd
|
||||
# Scheduled failures are tolerated only during the initial stabilization window.
|
||||
# The enforcement step below makes manual runs and later schedules blocking.
|
||||
continue-on-error: true
|
||||
shell: bash
|
||||
run: |
|
||||
set +e
|
||||
set -o pipefail
|
||||
DOCKER_API_VERSION=1.48 E2E_KEEP_LOGS=true \
|
||||
GINKGO_ARGS="--json-report=report.json --output-dir=/tmp/kube-vip-etcd-artifacts" \
|
||||
make e2e-tests-etcd 2>&1 | tee /tmp/kube-vip-etcd-artifacts/suite.log
|
||||
exit_code=${PIPESTATUS[0]}
|
||||
echo "exit_code=$exit_code" >> "$GITHUB_OUTPUT"
|
||||
exit "$exit_code"
|
||||
- name: Summarize Etcd suite
|
||||
if: always()
|
||||
env:
|
||||
OUTCOME: ${{ steps.etcd.outcome }}
|
||||
EXIT_CODE: ${{ steps.etcd.outputs.exit_code }}
|
||||
run: |
|
||||
echo "### Etcd E2E result: ${OUTCOME}" >> "$GITHUB_STEP_SUMMARY"
|
||||
printf '{"outcome":"%s","exit_code":%s,"event":"%s","cutoff":"2026-10-01"}\n' \
|
||||
"${OUTCOME:-skipped}" "${EXIT_CODE:-null}" "$GITHUB_EVENT_NAME" \
|
||||
> /tmp/kube-vip-etcd-artifacts/result.json
|
||||
- name: Save logs
|
||||
uses: actions/upload-artifact@v7
|
||||
continue-on-error: true
|
||||
with:
|
||||
name: etcd-e2e-logs
|
||||
path: |
|
||||
/tmp/kube-vip-etcd-artifacts
|
||||
/tmp/kube-vip-test*
|
||||
if-no-files-found: warn
|
||||
if: always()
|
||||
- name: Enforce Etcd result
|
||||
if: always()
|
||||
env:
|
||||
EVENT_NAME: ${{ github.event_name }}
|
||||
OUTCOME: ${{ steps.etcd.outcome }}
|
||||
run: |
|
||||
if test "$OUTCOME" = success; then
|
||||
exit 0
|
||||
fi
|
||||
if test "$EVENT_NAME" = schedule && test "$(date -u +%Y-%m-%d)" \< 2026-10-01; then
|
||||
echo "::warning::etcd e2e suite outcome was ${OUTCOME:-skipped} during stabilization through 2026-09-30"
|
||||
exit 0
|
||||
fi
|
||||
echo "::error::etcd e2e suite outcome was ${OUTCOME:-skipped}; see the etcd-e2e-logs artifact"
|
||||
exit 1
|
||||
1
.gitignore
vendored
1
.gitignore
vendored
@@ -3,6 +3,7 @@ kube-vip
|
||||
.vscode
|
||||
bin
|
||||
testing/e2e/etcd/certs
|
||||
coverage.out
|
||||
pkg/etcd/etcd.pid
|
||||
pkg/etcd/etcd-data
|
||||
testing/e2e/e2e.test
|
||||
|
||||
@@ -8,6 +8,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
- Propagate `bgp_attach_ip_to_interface` into per-service config so it attaches BGP-mode Service VIPs to the interface as configured.
|
||||
- Add a configurable kube-vip instance name and use it to isolate internal nftables egress tables, persist table ownership on Services, and migrate per-Service chains without affecting other deployments. Fixes #1634.
|
||||
- Retry on 403 Forbidden and 401 Unauthorized in `ServicesWatcher` at startup with exponential backoff. Fixes #1464.
|
||||
- Reintroduce BGP config via node annotations. Fixes #1488.
|
||||
@@ -50,6 +51,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
- Added support in ipoib interfaces in ARP mode. Fixes #694
|
||||
|
||||
### Changed
|
||||
- BGP mode now honours `enable_leader_election` for services: a single global services leader advertises the service VIPs instead of every node advertising them. Deployments that enabled `enable_leader_election` for the control plane and relied on ECMP/multipath for services must unset it (or switch to `enable_service_election`) to keep the previous datapath. kube-vip logs a warning on startup when this path is taken.
|
||||
- Updated signal handlers in manager_arp.go, manager_bgp.go, manager_wireguard.go, and manager_table.go to use switch statement pattern for handling multiple signals (SIGUSR1, SIGINT, SIGTERM)
|
||||
- wireguard.go now manages a complete wireguard interface on the current network namespace
|
||||
- manager_wireguard.go uses the new wireguard.go implementation
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# syntax=docker/dockerfile:experimental
|
||||
|
||||
FROM golang:1.26.5-alpine3.23 as dev
|
||||
FROM golang:1.27.1-alpine3.23 as dev
|
||||
RUN apk add --no-cache git ca-certificates make
|
||||
RUN adduser -D appuser
|
||||
COPY . /src/
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# syntax=docker/dockerfile:experimental
|
||||
|
||||
FROM golang:1.26.5-alpine3.23 as dev
|
||||
FROM golang:1.27.1-alpine3.23 as dev
|
||||
RUN apk add --no-cache git make
|
||||
RUN adduser -D appuser
|
||||
COPY . /src/
|
||||
|
||||
31
Makefile
31
Makefile
@@ -5,7 +5,7 @@ TARGET := kube-vip
|
||||
.DEFAULT_GOAL := $(TARGET)
|
||||
|
||||
# These will be provided to the target
|
||||
VERSION := v1.2.1
|
||||
VERSION := v1.2.3
|
||||
|
||||
BUILD := `git rev-parse HEAD`
|
||||
|
||||
@@ -16,10 +16,14 @@ TARGETOS=linux
|
||||
LDFLAGS=-ldflags "-s -w -X=main.Version=$(VERSION) -X=main.Build=$(BUILD) -extldflags -static"
|
||||
DOCKERTAG ?= $(VERSION)
|
||||
REPOSITORY ?= docker.io/plndr
|
||||
GO_VERSION := 1.25.6
|
||||
GO_VERSION := $(word 2,$(shell grep '^go ' go.mod))
|
||||
K8S_VERSION ?= v1.35.0
|
||||
GINKGO_ARGS ?=
|
||||
GINKGO_PROCS ?=
|
||||
GINKGO_PARALLEL := $(if $(GINKGO_PROCS),--procs=$(GINKGO_PROCS),-p)
|
||||
BUILDX_CACHE_FLAGS ?=
|
||||
|
||||
.PHONY: all build clean install uninstall simplify check run e2e-tests unit-tests integration-tests unit-tests-docker integration-tests-docker
|
||||
.PHONY: all build clean install uninstall simplify check run e2e-tests unit-tests integration-tests unit-tests-docker integration-tests-docker e2e-tests-etcd
|
||||
|
||||
all: check install
|
||||
|
||||
@@ -77,17 +81,17 @@ docker:
|
||||
# This will build a local docker image (x86 only), use make dockerLocal for all architectures
|
||||
dockerx86Local:
|
||||
@-rm ./kube-vip
|
||||
@docker buildx build --platform linux/amd64 --load -t $(REPOSITORY)/$(TARGET):$(DOCKERTAG) .
|
||||
@docker buildx build --platform linux/amd64 --load -t $(REPOSITORY)/$(TARGET):$(DOCKERTAG) $(BUILDX_CACHE_FLAGS) .
|
||||
@echo New Multi Architecture Docker image created
|
||||
|
||||
dockerx86Action:
|
||||
@-rm ./kube-vip
|
||||
@docker buildx build --platform linux/amd64 --load -t $(REPOSITORY)/$(TARGET):action .
|
||||
@docker buildx build --platform linux/amd64 --load -t $(REPOSITORY)/$(TARGET):action $(BUILDX_CACHE_FLAGS) .
|
||||
@echo New Multi Architecture Docker image created
|
||||
|
||||
dockerx86ActionIPTables:
|
||||
@-rm ./kube-vip
|
||||
@docker buildx build --platform linux/amd64 -f ./Dockerfile_iptables --load -t $(REPOSITORY)/$(TARGET):action .
|
||||
@docker buildx build --platform linux/amd64 -f ./Dockerfile_iptables --load -t $(REPOSITORY)/$(TARGET):action $(BUILDX_CACHE_FLAGS) .
|
||||
@echo New Multi Architecture Docker image created
|
||||
|
||||
dockerLocal:
|
||||
@@ -129,28 +133,31 @@ manifest-test:
|
||||
docker run $(REPOSITORY)/$(TARGET):$(DOCKERTAG) manifest daemonset --interface eth0 --vip 192.168.0.1 --image "$(REPOSITORY)/$(TARGET):$(DOCKERTAG)" --bgp --leaderElection --controlplane --services --inCluster
|
||||
|
||||
unit-tests:
|
||||
go test -race ./...
|
||||
go test -race -coverprofile=coverage.out -covermode=atomic ./...
|
||||
|
||||
unit-tests-docker:
|
||||
docker run --rm -w /kube-vip -v $$(pwd):/kube-vip -v kube-vip-gomod-cache:/go/pkg/mod -v kube-vip-gobuild-cache:/root/.cache/go-build golang:$(GO_VERSION) make unit-tests
|
||||
docker run --rm -w /kube-vip -v $$(pwd):/kube-vip -v kube-vip-gomod-cache:/go/pkg/mod -v kube-vip-gobuild-cache:/root/.cache/go-build golang:$(GO_VERSION) sh -c "make unit-tests; status=$$?; chmod 666 coverage.out 2>/dev/null || true; exit $$status"
|
||||
|
||||
integration-tests:
|
||||
go test -tags=integration,e2e -v ./pkg/etcd
|
||||
|
||||
e2e-tests-arp: get-whoami
|
||||
GOMAXPROCS=4 TEST_MODE=arp K8S_IMAGE_PATH=kindest/node:$(K8S_VERSION) E2E_IMAGE_PATH=$(REPOSITORY)/$(TARGET):$(DOCKERTAG) go run github.com/onsi/ginkgo/v2/ginkgo --tags=e2e -v -p ./testing/e2e
|
||||
GOMAXPROCS=4 TEST_MODE=arp K8S_IMAGE_PATH=kindest/node:$(K8S_VERSION) E2E_IMAGE_PATH=$(REPOSITORY)/$(TARGET):$(DOCKERTAG) go run github.com/onsi/ginkgo/v2/ginkgo --tags=e2e -v $(GINKGO_PARALLEL) $(GINKGO_ARGS) ./testing/e2e
|
||||
|
||||
e2e-tests-rt: get-whoami
|
||||
GOMAXPROCS=4 TEST_MODE=rt K8S_IMAGE_PATH=kindest/node:$(K8S_VERSION) E2E_IMAGE_PATH=$(REPOSITORY)/$(TARGET):$(DOCKERTAG) go run github.com/onsi/ginkgo/v2/ginkgo --tags=e2e -v -p ./testing/e2e
|
||||
GOMAXPROCS=4 TEST_MODE=rt K8S_IMAGE_PATH=kindest/node:$(K8S_VERSION) E2E_IMAGE_PATH=$(REPOSITORY)/$(TARGET):$(DOCKERTAG) go run github.com/onsi/ginkgo/v2/ginkgo --tags=e2e -v $(GINKGO_PARALLEL) $(GINKGO_ARGS) ./testing/e2e
|
||||
|
||||
e2e-tests-bgp: get-whoami get-gobgp
|
||||
GOMAXPROCS=4 TEST_MODE=bgp K8S_IMAGE_PATH=kindest/node:$(K8S_VERSION) E2E_IMAGE_PATH=$(REPOSITORY)/$(TARGET):$(DOCKERTAG) go run github.com/onsi/ginkgo/v2/ginkgo --tags=e2e -v -p ./testing/e2e
|
||||
GOMAXPROCS=4 TEST_MODE=bgp K8S_IMAGE_PATH=kindest/node:$(K8S_VERSION) E2E_IMAGE_PATH=$(REPOSITORY)/$(TARGET):$(DOCKERTAG) go run github.com/onsi/ginkgo/v2/ginkgo --tags=e2e -v $(GINKGO_PARALLEL) $(GINKGO_ARGS) ./testing/e2e
|
||||
|
||||
e2e-tests-etcd: get-whoami
|
||||
GOMAXPROCS=4 K8S_IMAGE_PATH=kindest/node:$(K8S_VERSION) E2E_IMAGE_PATH=$(REPOSITORY)/$(TARGET):$(DOCKERTAG) go run github.com/onsi/ginkgo/v2/ginkgo --tags=e2e -v $(GINKGO_PARALLEL) $(GINKGO_ARGS) ./testing/e2e/etcd
|
||||
|
||||
e2e-tests: e2e-tests-arp e2e-tests-rt e2e-tests-bgp
|
||||
|
||||
service-tests:
|
||||
$(MAKE) -C testing/e2e/e2e dockerLocal
|
||||
E2E_IMAGE_PATH=$(REPOSITORY)/$(TARGET):$(DOCKERTAG) go run ./testing/services -Services -simple -deployments -leaderActive -leaderFailover -localDeploy -egress -egressIPv6 -dualStack -egressInternal
|
||||
E2E_IMAGE_PATH=$(REPOSITORY)/$(TARGET):$(DOCKERTAG) go run ./testing/services -Services -simple -deployments -leaderActive -leaderFailover -localDeploy -electionFaults -egress -egressIPv6 -dualStack -egressInternal
|
||||
|
||||
trivy: dockerx86ActionIPTables
|
||||
docker run -v /var/run/docker.sock:/var/run/docker.sock aquasec/trivy:0.47.0 \
|
||||
|
||||
@@ -18,6 +18,8 @@ The idea behind `kube-vip` is a small self-contained Highly-Available option for
|
||||
|
||||
**NOTE:** All documentation of both usage and architecture are now available at [https://kube-vip.io](https://kube-vip.io).
|
||||
|
||||
For upgrading an existing install in place (static Pod or DaemonSet), see the [upgrade guide](https://kube-vip.io/docs/upgrade/).
|
||||
|
||||
## Features
|
||||
|
||||
Kube-Vip was originally created to provide a HA solution for the Kubernetes control plane, over time it has evolved to incorporate that same functionality into Kubernetes service type [load-balancers](https://kubernetes.io/docs/concepts/services-networking/service/#loadbalancer).
|
||||
@@ -130,7 +132,7 @@ Additionally it is now relatively easy and quick to develop with [skaffold](http
|
||||
|
||||
## Star History
|
||||
|
||||
[](https://star-history.com/#kube-vip/kube-vip&Date)
|
||||
[](https://star-history.dera.page/#kube-vip/kube-vip&type=date)
|
||||
[](https://app.fossa.com/projects/git%2Bgithub.com%2Fkube-vip%2Fkube-vip?ref=badge_shield)
|
||||
|
||||
|
||||
|
||||
110
cmd/kube-vip.go
110
cmd/kube-vip.go
@@ -52,8 +52,9 @@ var (
|
||||
)
|
||||
|
||||
var kubeVipCmd = &cobra.Command{
|
||||
Use: "kube-vip",
|
||||
Short: "This is a server for providing a Virtual IP and load-balancer for the Kubernetes control-plane",
|
||||
Use: "kube-vip",
|
||||
Short: "This is a server for providing a Virtual IP and load-balancer for the Kubernetes control-plane",
|
||||
SilenceErrors: true,
|
||||
}
|
||||
|
||||
func init() {
|
||||
@@ -92,6 +93,7 @@ func init() {
|
||||
|
||||
// BGP flags
|
||||
kubeVipCmd.PersistentFlags().BoolVar(&initConfig.EnableBGP, "bgp", false, "This will enable BGP support within kube-vip")
|
||||
kubeVipCmd.PersistentFlags().BoolVar(&initConfig.BGPAttachIPToInterface, "bgpAttachIPToInterface", false, "Assign BGP service VIPs to the configured interface")
|
||||
kubeVipCmd.PersistentFlags().StringVar(&initConfig.BGPConfig.RouterID, "bgpRouterID", "", "The routerID for the bgp server")
|
||||
kubeVipCmd.PersistentFlags().StringVar(&initConfig.BGPConfig.SourceIF, "sourceIF", "", "The source interface for bgp peering (not to be used with sourceIP)")
|
||||
kubeVipCmd.PersistentFlags().StringVar(&initConfig.BGPConfig.SourceIP, "sourceIP", "", "The source address for bgp peering (not to be used with sourceIF)")
|
||||
@@ -186,11 +188,16 @@ func init() {
|
||||
}
|
||||
|
||||
// Execute - starts the command parsing process
|
||||
func Execute() {
|
||||
if err := kubeVipCmd.Execute(); err != nil {
|
||||
fmt.Println(err)
|
||||
os.Exit(1)
|
||||
func Execute() int {
|
||||
cmd, err := kubeVipCmd.ExecuteC()
|
||||
if err != nil {
|
||||
log.Error("command failed", "err", err)
|
||||
if cmd == kubeVipCmd {
|
||||
_ = cmd.Usage()
|
||||
}
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
var kubeVipVersion = &cobra.Command{
|
||||
@@ -214,26 +221,24 @@ var kubeVipSample = &cobra.Command{
|
||||
var kubeVipService = &cobra.Command{
|
||||
Use: "service",
|
||||
Short: "Start the Virtual IP / Load balancer as a service within a Kubernetes cluster",
|
||||
Run: func(cmd *cobra.Command, args []string) { //nolint TODO
|
||||
RunE: func(cmd *cobra.Command, args []string) error { //nolint TODO
|
||||
cmd.SilenceUsage = true
|
||||
|
||||
// Load configuration from file if specified (lowest priority)
|
||||
if initConfig.ConfigFile != "" {
|
||||
err := kubevip.MergeConfigFromFile(&initConfig, initConfig.ConfigFile)
|
||||
if err != nil {
|
||||
log.Error("loading config file", "err", err)
|
||||
return
|
||||
return fmt.Errorf("loading config file: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
// parse environment variables, these will overwrite anything loaded from config file
|
||||
err := kubevip.ParseEnvironment(&initConfig)
|
||||
if err != nil {
|
||||
log.Error("parsing env", "err", err)
|
||||
return
|
||||
return fmt.Errorf("parsing environment: %w", err)
|
||||
}
|
||||
if err := initConfig.Validate(); err != nil {
|
||||
log.Error("validating configuration", "err", err)
|
||||
return
|
||||
return fmt.Errorf("validating configuration: %w", err)
|
||||
}
|
||||
|
||||
// Change RTN_UNSPEC to default type
|
||||
@@ -245,8 +250,7 @@ var kubeVipService = &cobra.Command{
|
||||
log.SetLogLoggerLevel(log.Level(initConfig.Logging))
|
||||
|
||||
if err := initConfig.CheckInterface(); err != nil {
|
||||
log.Error("checking interface", "err", err)
|
||||
return
|
||||
return fmt.Errorf("checking interface: %w", err)
|
||||
}
|
||||
|
||||
// User Environment variables as an option to make manifest clearer
|
||||
@@ -259,8 +263,7 @@ var kubeVipService = &cobra.Command{
|
||||
if initConfig.EnableControlPlane &&
|
||||
(initConfig.EnableARP || initConfig.EnableBGP || initConfig.EnableRoutingTable) {
|
||||
if err := initConfig.CheckSubnetExists(); err != nil {
|
||||
log.Error("checking subnet exists if vip_address defined", "err", err)
|
||||
return
|
||||
return fmt.Errorf("checking subnet exists if vip_address defined: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -268,8 +271,7 @@ var kubeVipService = &cobra.Command{
|
||||
if initConfig.VIPSubnet == "" && initConfig.Address != "" {
|
||||
initConfig.VIPSubnet, err = GenerateCidrRange(initConfig.Address, initConfig.DNSMode)
|
||||
if err != nil {
|
||||
log.Error("generating CIDR", "err", err)
|
||||
return
|
||||
return fmt.Errorf("generating CIDR: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -279,41 +281,39 @@ var kubeVipService = &cobra.Command{
|
||||
// Define the new service manager
|
||||
mgr, err := manager.New(ctx, configMap, &initConfig)
|
||||
if err != nil {
|
||||
log.Error("new manager", "err", err)
|
||||
return
|
||||
return fmt.Errorf("new manager: %w", err)
|
||||
}
|
||||
|
||||
// Start the service manager, this will watch the config Map and construct kube-vip services for it
|
||||
err = mgr.Start(ctx)
|
||||
if err != nil {
|
||||
log.Error("manager start", "err", err)
|
||||
return
|
||||
return fmt.Errorf("manager start: %w", err)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
}
|
||||
|
||||
var kubeVipManager = &cobra.Command{
|
||||
Use: "manager",
|
||||
Short: "Start the kube-vip manager",
|
||||
Run: func(cmd *cobra.Command, args []string) { //nolint TODO
|
||||
RunE: func(cmd *cobra.Command, args []string) error { //nolint TODO
|
||||
cmd.SilenceUsage = true
|
||||
|
||||
// Load configuration from file if specified (lowest priority)
|
||||
if initConfig.ConfigFile != "" {
|
||||
err := kubevip.MergeConfigFromFile(&initConfig, initConfig.ConfigFile)
|
||||
if err != nil {
|
||||
log.Error("loading config file", "err", err)
|
||||
return
|
||||
return fmt.Errorf("loading config file: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
// parse environment variables, these will overwrite anything loaded from config file
|
||||
err := kubevip.ParseEnvironment(&initConfig)
|
||||
if err != nil {
|
||||
log.Error("parsing environment", "err", err)
|
||||
return
|
||||
return fmt.Errorf("parsing environment: %w", err)
|
||||
}
|
||||
if err := initConfig.Validate(); err != nil {
|
||||
log.Error("validating configuration", "err", err)
|
||||
return
|
||||
return fmt.Errorf("validating configuration: %w", err)
|
||||
}
|
||||
|
||||
// Change RTN_UNSPEC to default type
|
||||
@@ -328,8 +328,7 @@ var kubeVipManager = &cobra.Command{
|
||||
if initConfig.EnableControlPlane &&
|
||||
(initConfig.EnableARP || initConfig.EnableBGP || initConfig.EnableRoutingTable) {
|
||||
if err := initConfig.CheckSubnetExists(); err != nil {
|
||||
log.Error("checking subnet exists if vip_address defined", "err", err)
|
||||
return
|
||||
return fmt.Errorf("checking subnet exists if vip_address defined: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -337,8 +336,7 @@ var kubeVipManager = &cobra.Command{
|
||||
if initConfig.VIPSubnet == "" && initConfig.Address != "" {
|
||||
initConfig.VIPSubnet, err = GenerateCidrRange(initConfig.Address, initConfig.DNSMode)
|
||||
if err != nil {
|
||||
log.Error("No interface is specified for kube-vip to bind to")
|
||||
return
|
||||
return fmt.Errorf("generating CIDR: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -349,8 +347,8 @@ var kubeVipManager = &cobra.Command{
|
||||
defer wg.Wait()
|
||||
|
||||
// create main manager context
|
||||
ctx, cancel := context.WithCancel(cmd.Context())
|
||||
defer cancel()
|
||||
ctx, cancel := context.WithCancelCause(cmd.Context())
|
||||
defer cancel(nil)
|
||||
|
||||
// start prometheus server
|
||||
if initConfig.PrometheusHTTPServer != "" {
|
||||
@@ -387,13 +385,11 @@ var kubeVipManager = &cobra.Command{
|
||||
}
|
||||
|
||||
if mode == "" {
|
||||
log.Error("no valid kube-vip mode detected, ensure a supported mode is configured")
|
||||
return
|
||||
return fmt.Errorf("no valid kube-vip mode detected, ensure a supported mode is configured")
|
||||
}
|
||||
|
||||
if modesEnabled > 1 {
|
||||
log.Error("multiple kube-vip modes detected, ensure only one mode is configured")
|
||||
return
|
||||
return fmt.Errorf("multiple kube-vip modes detected, ensure only one mode is configured")
|
||||
}
|
||||
|
||||
// Provide configuration to output/logging
|
||||
@@ -401,18 +397,15 @@ var kubeVipManager = &cobra.Command{
|
||||
|
||||
// End if nothing is enabled
|
||||
if !initConfig.EnableServices && !initConfig.EnableControlPlane {
|
||||
log.Error("no features are enabled")
|
||||
return
|
||||
return fmt.Errorf("no features are enabled")
|
||||
}
|
||||
|
||||
if !initConfig.EnableARP && strings.Contains(initConfig.VIPSubnet, kubevip.Auto) {
|
||||
log.Error("auto subnet discovery cannot be used outside ARP mode")
|
||||
return
|
||||
return fmt.Errorf("auto subnet discovery cannot be used outside ARP mode")
|
||||
}
|
||||
|
||||
if strings.Contains(initConfig.VIPSubnet, kubevip.Auto) && initConfig.Address != "" {
|
||||
log.Error("auto subnet discovery cannot be used if VIP address was provided")
|
||||
return
|
||||
return fmt.Errorf("auto subnet discovery cannot be used if VIP address was provided")
|
||||
}
|
||||
|
||||
// If we're using wireguard then all traffic goes through the wg0 interface
|
||||
@@ -429,20 +422,17 @@ var kubeVipManager = &cobra.Command{
|
||||
log.Warn("attempting to create wireguard interface", "interface not found", initConfig.Interface)
|
||||
err = netlink.LinkAdd(&netlink.Wireguard{LinkAttrs: netlink.LinkAttrs{Name: initConfig.Interface}})
|
||||
if err != nil {
|
||||
log.Error("adding link", "err", err)
|
||||
return
|
||||
return fmt.Errorf("adding link: %w", err)
|
||||
}
|
||||
l, err = netlink.LinkByName(initConfig.Interface)
|
||||
if err != nil {
|
||||
log.Error("finding link", "err", err)
|
||||
return
|
||||
return fmt.Errorf("finding link: %w", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
err = netlink.LinkSetUp(l)
|
||||
if err != nil {
|
||||
log.Error("setting link UP", "err", err)
|
||||
return
|
||||
return fmt.Errorf("setting link UP: %w", err)
|
||||
}
|
||||
|
||||
} else { // if we're not using Wireguard then we'll need to use an actual interface
|
||||
@@ -452,25 +442,22 @@ var kubeVipManager = &cobra.Command{
|
||||
defaultIF, err := vip.GetDefaultGatewayInterface()
|
||||
if err != nil {
|
||||
_ = cmd.Help()
|
||||
log.Error("detecting interface", "err", err)
|
||||
return
|
||||
return fmt.Errorf("detecting interface: %w", err)
|
||||
}
|
||||
initConfig.Interface = defaultIF.Name
|
||||
log.Info("kube-vip bind", "interface", initConfig.Interface)
|
||||
|
||||
wg.Go(func() {
|
||||
if err := vip.MonitorDefaultInterface(ctx, defaultIF); err != nil {
|
||||
|
||||
log.Error("interface monitor", "err", err)
|
||||
return
|
||||
cancel(err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
// Perform a check on the state of the interface
|
||||
if err := initConfig.CheckInterface(); err != nil {
|
||||
log.Error("checking interface", "err", err)
|
||||
return
|
||||
return fmt.Errorf("checking interface: %w", err)
|
||||
}
|
||||
|
||||
// User Environment variables as an option to make manifest clearer
|
||||
@@ -482,8 +469,7 @@ var kubeVipManager = &cobra.Command{
|
||||
// Define the new service manager
|
||||
mgr, err := manager.New(ctx, configMap, &initConfig)
|
||||
if err != nil {
|
||||
log.Error("new manager", "err", err)
|
||||
return
|
||||
return fmt.Errorf("new manager: %w", err)
|
||||
}
|
||||
|
||||
metrics.RegisterPrometheusMetrics()
|
||||
@@ -492,9 +478,9 @@ var kubeVipManager = &cobra.Command{
|
||||
// Start the service manager, this will watch the config Map and construct kube-vip services for it
|
||||
err = mgr.Start(ctx)
|
||||
if err != nil {
|
||||
log.Error("start manager", "err", err)
|
||||
return
|
||||
return fmt.Errorf("start manager: %w", err)
|
||||
}
|
||||
return nil
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@@ -1,74 +0,0 @@
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
app: kube-vip-cluster
|
||||
name: kube-vip-cluster
|
||||
spec:
|
||||
replicas: 3
|
||||
selector:
|
||||
matchLabels:
|
||||
app: kube-vip-cluster
|
||||
strategy: {}
|
||||
template:
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
app: kube-vip-cluster
|
||||
spec:
|
||||
affinity:
|
||||
podAntiAffinity:
|
||||
requiredDuringSchedulingIgnoredDuringExecution:
|
||||
- labelSelector:
|
||||
matchExpressions:
|
||||
- key: "app"
|
||||
operator: In
|
||||
values:
|
||||
- kube-vip-cluster
|
||||
topologyKey: "kubernetes.io/hostname"
|
||||
containers:
|
||||
- image: ghcr.io/kube-vip/kube-vip:0.3.7
|
||||
imagePullPolicy: Always
|
||||
name: kube-vip
|
||||
command:
|
||||
- /kube-vip
|
||||
- service
|
||||
- --configMap
|
||||
- plndr-configmap
|
||||
- --arp
|
||||
- --interface
|
||||
- ens192
|
||||
- --log
|
||||
- "5"
|
||||
resources: {}
|
||||
securityContext:
|
||||
capabilities:
|
||||
add:
|
||||
- NET_ADMIN
|
||||
hostNetwork: true
|
||||
status: {}
|
||||
---
|
||||
kind: Role
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
metadata:
|
||||
name: lease-access
|
||||
rules:
|
||||
- apiGroups: ["coordination.k8s.io"]
|
||||
resources: ["leases"]
|
||||
verbs: ["get", "create", "update", "list", "put"]
|
||||
- apiGroups: [""]
|
||||
resources: ["configMap"]
|
||||
verbs: ["get"]
|
||||
---
|
||||
kind: RoleBinding
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
metadata:
|
||||
name: lease-access
|
||||
subjects:
|
||||
- kind: User
|
||||
name: system:serviceaccount:default:default
|
||||
roleRef:
|
||||
kind: Role
|
||||
name: lease-access
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
@@ -1,83 +0,0 @@
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: vip
|
||||
---
|
||||
kind: Role
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
metadata:
|
||||
name: vip-role
|
||||
rules:
|
||||
- apiGroups: ["coordination.k8s.io"]
|
||||
resources: ["leases"]
|
||||
verbs: ["get", "create", "update", "list", "put"]
|
||||
- apiGroups: [""]
|
||||
resources: ["configmaps", "endpoints"]
|
||||
verbs: ["watch", "get"]
|
||||
---
|
||||
kind: RoleBinding
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
metadata:
|
||||
name: vip-role-bind
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: vip
|
||||
apiGroup: ""
|
||||
roleRef:
|
||||
kind: Role
|
||||
name: vip-role
|
||||
apiGroup: ""
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
app: kube-vip-cluster
|
||||
name: kube-vip-cluster
|
||||
spec:
|
||||
replicas: 3
|
||||
selector:
|
||||
matchLabels:
|
||||
app: kube-vip-cluster
|
||||
strategy: {}
|
||||
template:
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
app: kube-vip-cluster
|
||||
spec:
|
||||
affinity:
|
||||
podAntiAffinity:
|
||||
requiredDuringSchedulingIgnoredDuringExecution:
|
||||
- labelSelector:
|
||||
matchExpressions:
|
||||
- key: "app"
|
||||
operator: In
|
||||
values:
|
||||
- kube-vip-cluster
|
||||
topologyKey: "kubernetes.io/hostname"
|
||||
containers:
|
||||
- image: ghcr.io/kube-vip/kube-vip:0.3.7
|
||||
imagePullPolicy: Always
|
||||
name: kube-vip
|
||||
command:
|
||||
- /kube-vip
|
||||
- service
|
||||
env:
|
||||
- name: vip_interface
|
||||
value: "ens192"
|
||||
- name: vip_configmap
|
||||
value: "plndr"
|
||||
- name: vip_arp
|
||||
value: "true"
|
||||
- name: vip_loglevel
|
||||
value: "5"
|
||||
resources: {}
|
||||
securityContext:
|
||||
capabilities:
|
||||
add:
|
||||
- NET_ADMIN
|
||||
hostNetwork: true
|
||||
serviceAccountName: vip
|
||||
status: {}
|
||||
@@ -1,83 +0,0 @@
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: vip
|
||||
---
|
||||
kind: Role
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
metadata:
|
||||
name: vip-role
|
||||
rules:
|
||||
- apiGroups: ["coordination.k8s.io"]
|
||||
resources: ["leases"]
|
||||
verbs: ["get", "create", "update", "list", "put"]
|
||||
- apiGroups: [""]
|
||||
resources: ["configmaps", "endpoints"]
|
||||
verbs: ["watch", "get"]
|
||||
---
|
||||
kind: RoleBinding
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
metadata:
|
||||
name: vip-role-bind
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: vip
|
||||
apiGroup: ""
|
||||
roleRef:
|
||||
kind: Role
|
||||
name: vip-role
|
||||
apiGroup: ""
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
app: kube-vip-cluster
|
||||
name: kube-vip-cluster
|
||||
spec:
|
||||
replicas: 3
|
||||
selector:
|
||||
matchLabels:
|
||||
app: kube-vip-cluster
|
||||
strategy: {}
|
||||
template:
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
app: kube-vip-cluster
|
||||
spec:
|
||||
affinity:
|
||||
podAntiAffinity:
|
||||
requiredDuringSchedulingIgnoredDuringExecution:
|
||||
- labelSelector:
|
||||
matchExpressions:
|
||||
- key: "app"
|
||||
operator: In
|
||||
values:
|
||||
- kube-vip-cluster
|
||||
topologyKey: "kubernetes.io/hostname"
|
||||
containers:
|
||||
- image: plndr/kube-vip:0.1.4
|
||||
imagePullPolicy: Always
|
||||
name: kube-vip
|
||||
command:
|
||||
- /kube-vip
|
||||
- service
|
||||
env:
|
||||
- name: vip_interface
|
||||
value: "ens192"
|
||||
- name: vip_configmap
|
||||
value: "plndr"
|
||||
- name: vip_arp
|
||||
value: "true"
|
||||
- name: vip_loglevel
|
||||
value: "5"
|
||||
resources: {}
|
||||
securityContext:
|
||||
capabilities:
|
||||
add:
|
||||
- NET_ADMIN
|
||||
hostNetwork: true
|
||||
serviceAccountName: vip
|
||||
status: {}
|
||||
@@ -1,55 +0,0 @@
|
||||
apiVersion: apps/v1
|
||||
kind: DaemonSet
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
name: kube-vip-ds
|
||||
namespace: kube-system
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
name: kube-vip-ds
|
||||
template:
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
name: kube-vip-ds
|
||||
spec:
|
||||
containers:
|
||||
- args:
|
||||
- manager
|
||||
env:
|
||||
- name: vip_arp
|
||||
value: "true"
|
||||
- name: vip_interface
|
||||
value: eth0
|
||||
- name: port
|
||||
value: "6443"
|
||||
- name: vip_cidr
|
||||
value: "32"
|
||||
- name: svc_enable
|
||||
value: "true"
|
||||
- name: vip_startleader
|
||||
value: "false"
|
||||
- name: vip_addpeerstolb
|
||||
value: "true"
|
||||
- name: vip_localpeer
|
||||
value: ip-172-20-40-207:172.20.40.207:10000
|
||||
- name: vip_address
|
||||
image: plndr/kube-vip:v0.3.5
|
||||
imagePullPolicy: Always
|
||||
name: kube-vip
|
||||
resources: {}
|
||||
securityContext:
|
||||
capabilities:
|
||||
add:
|
||||
- NET_ADMIN
|
||||
- NET_RAW
|
||||
- SYS_TIME
|
||||
hostNetwork: true
|
||||
serviceAccountName: kube-vip
|
||||
updateStrategy: {}
|
||||
status:
|
||||
currentNumberScheduled: 0
|
||||
desiredNumberScheduled: 0
|
||||
numberMisscheduled: 0
|
||||
numberReady: 0
|
||||
80
go.mod
80
go.mod
@@ -4,41 +4,41 @@ go 1.26.4
|
||||
|
||||
require (
|
||||
github.com/cloudflare/ipvs v0.12.0
|
||||
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc
|
||||
github.com/containernetworking/plugins v1.9.1
|
||||
github.com/docker/docker v28.5.2+incompatible
|
||||
github.com/florianl/go-conntrack v0.7.0
|
||||
github.com/google/go-cmp v0.7.0
|
||||
github.com/google/go-containerregistry v0.21.7
|
||||
github.com/google/go-containerregistry v0.22.1
|
||||
github.com/google/nftables v0.3.0
|
||||
github.com/gookit/slog v0.7.1
|
||||
github.com/huin/goupnp v1.3.0
|
||||
github.com/insomniacslk/dhcp v0.0.0-20241224095048-b56fa0d5f25d
|
||||
github.com/insomniacslk/dhcp v0.0.0-20260719225207-c76316d4aa82
|
||||
github.com/jpillora/backoff v1.0.0
|
||||
github.com/mdlayher/ndp v1.1.0
|
||||
github.com/onsi/ginkgo/v2 v2.32.0
|
||||
github.com/onsi/gomega v1.42.1
|
||||
github.com/osrg/gobgp/v4 v4.7.0
|
||||
github.com/onsi/ginkgo/v2 v2.32.2
|
||||
github.com/onsi/gomega v1.43.0
|
||||
github.com/osrg/gobgp/v4 v4.9.0
|
||||
github.com/pkg/errors v0.9.1
|
||||
github.com/prometheus/client_golang v1.23.2
|
||||
github.com/sirupsen/logrus v1.9.4
|
||||
github.com/prometheus/client_golang v1.24.1
|
||||
github.com/spf13/cobra v1.10.2
|
||||
github.com/stretchr/testify v1.11.1
|
||||
github.com/vishvananda/netlink v1.3.1
|
||||
go.etcd.io/etcd/api/v3 v3.6.12
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.13
|
||||
go.etcd.io/etcd/client/v3 v3.6.12
|
||||
github.com/stretchr/testify v1.12.1
|
||||
github.com/vishvananda/netlink v1.3.2-0.20260830232854-cf01b55a4a4b
|
||||
github.com/vishvananda/netns v0.0.5
|
||||
go.etcd.io/etcd/api/v3 v3.7.1
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.7.1
|
||||
go.etcd.io/etcd/client/v3 v3.7.1
|
||||
go.uber.org/zap v1.28.0
|
||||
golang.org/x/exp v0.0.0-20250103183323-7d7fa50e5329
|
||||
golang.org/x/sync v0.22.0
|
||||
golang.org/x/sys v0.47.0
|
||||
golang.org/x/sync v0.23.0
|
||||
golang.org/x/sys v0.48.0
|
||||
golang.zx2c4.com/wireguard/wgctrl v0.0.0-20241231184526-a9ab2273dd10
|
||||
google.golang.org/grpc v1.82.1
|
||||
google.golang.org/grpc v1.83.2
|
||||
gopkg.in/yaml.v3 v3.0.1
|
||||
k8s.io/api v0.36.2
|
||||
k8s.io/apimachinery v0.36.2
|
||||
k8s.io/client-go v0.36.2
|
||||
k8s.io/api v0.36.4
|
||||
k8s.io/apimachinery v0.36.4
|
||||
k8s.io/client-go v0.36.4
|
||||
k8s.io/klog/v2 v2.140.0
|
||||
sigs.k8s.io/kind v0.32.0
|
||||
sigs.k8s.io/kind v0.33.0
|
||||
sigs.k8s.io/yaml v1.6.0
|
||||
)
|
||||
|
||||
@@ -53,7 +53,8 @@ require (
|
||||
github.com/containerd/errdefs/pkg v0.3.0 // indirect
|
||||
github.com/containerd/log v0.1.0 // indirect
|
||||
github.com/coreos/go-semver v0.3.1 // indirect
|
||||
github.com/coreos/go-systemd/v22 v22.5.0 // indirect
|
||||
github.com/coreos/go-systemd/v22 v22.7.0 // indirect
|
||||
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc // indirect
|
||||
github.com/dgryski/go-farm v0.0.0-20240924180020-3414d57e47da // indirect
|
||||
github.com/distribution/reference v0.6.0 // indirect
|
||||
github.com/docker/go-connections v0.7.0 // indirect
|
||||
@@ -73,7 +74,6 @@ require (
|
||||
github.com/go-openapi/swag v0.23.0 // indirect
|
||||
github.com/go-task/slim-sprig/v3 v3.0.0 // indirect
|
||||
github.com/go-viper/mapstructure/v2 v2.4.0 // indirect
|
||||
github.com/gogo/protobuf v1.3.2 // indirect
|
||||
github.com/golang/protobuf v1.5.4 // indirect
|
||||
github.com/google/gnostic-models v0.7.0 // indirect
|
||||
github.com/google/pprof v0.0.0-20260402051712-545e8a4df936 // indirect
|
||||
@@ -81,12 +81,13 @@ require (
|
||||
github.com/gookit/color v1.6.1 // indirect
|
||||
github.com/gookit/goutil v0.7.6 // indirect
|
||||
github.com/gookit/gsr v0.1.1 // indirect
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.28.0 // indirect
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.29.0 // indirect
|
||||
github.com/inconshreveable/mousetrap v1.1.0 // indirect
|
||||
github.com/josharian/intern v1.0.0 // indirect
|
||||
github.com/josharian/native v1.1.0 // indirect
|
||||
github.com/json-iterator/go v1.1.12 // indirect
|
||||
github.com/k-sone/critbitgo v1.4.0 // indirect
|
||||
github.com/kylelemons/godebug v1.1.0 // indirect
|
||||
github.com/mailru/easyjson v0.9.0 // indirect
|
||||
github.com/mattn/go-isatty v0.0.20 // indirect
|
||||
github.com/mdlayher/genetlink v1.3.2 // indirect
|
||||
@@ -108,8 +109,8 @@ require (
|
||||
github.com/pierrec/lz4/v4 v4.1.22 // indirect
|
||||
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 // indirect
|
||||
github.com/prometheus/client_model v0.6.2 // indirect
|
||||
github.com/prometheus/common v0.66.1 // indirect
|
||||
github.com/prometheus/procfs v0.16.1 // indirect
|
||||
github.com/prometheus/common v0.70.1 // indirect
|
||||
github.com/prometheus/procfs v0.21.1 // indirect
|
||||
github.com/sagikazarmark/locafero v0.7.0 // indirect
|
||||
github.com/segmentio/fasthash v1.0.3 // indirect
|
||||
github.com/sourcegraph/conc v0.3.0 // indirect
|
||||
@@ -120,29 +121,28 @@ require (
|
||||
github.com/subosito/gotenv v1.6.0 // indirect
|
||||
github.com/u-root/uio v0.0.0-20240224005618-d2acac8f3701 // indirect
|
||||
github.com/valyala/bytebufferpool v1.0.0 // indirect
|
||||
github.com/vishvananda/netns v0.0.5 // indirect
|
||||
github.com/x448/float16 v0.8.4 // indirect
|
||||
github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e // indirect
|
||||
go.opentelemetry.io/auto/sdk v1.2.1 // indirect
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.61.0 // indirect
|
||||
go.opentelemetry.io/otel v1.43.0 // indirect
|
||||
go.opentelemetry.io/otel v1.44.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.43.0 // indirect
|
||||
go.opentelemetry.io/otel/metric v1.43.0 // indirect
|
||||
go.opentelemetry.io/otel/trace v1.43.0 // indirect
|
||||
go.opentelemetry.io/otel/metric v1.44.0 // indirect
|
||||
go.opentelemetry.io/otel/trace v1.44.0 // indirect
|
||||
go.uber.org/multierr v1.11.0 // indirect
|
||||
go.yaml.in/yaml/v2 v2.4.3 // indirect
|
||||
go.yaml.in/yaml/v3 v3.0.4 // indirect
|
||||
golang.org/x/crypto v0.53.0 // indirect
|
||||
golang.org/x/mod v0.37.0 // indirect
|
||||
golang.org/x/net v0.56.0 // indirect
|
||||
go.yaml.in/yaml/v2 v2.4.4 // indirect
|
||||
go.yaml.in/yaml/v3 v3.0.5 // indirect
|
||||
golang.org/x/crypto v0.55.0 // indirect
|
||||
golang.org/x/mod v0.39.0 // indirect
|
||||
golang.org/x/net v0.58.0 // indirect
|
||||
golang.org/x/oauth2 v0.36.0 // indirect
|
||||
golang.org/x/term v0.44.0 // indirect
|
||||
golang.org/x/text v0.38.0 // indirect
|
||||
golang.org/x/term v0.45.0 // indirect
|
||||
golang.org/x/text v0.41.0 // indirect
|
||||
golang.org/x/time v0.14.0 // indirect
|
||||
golang.org/x/tools v0.46.0 // indirect
|
||||
golang.org/x/tools v0.49.0 // indirect
|
||||
golang.zx2c4.com/wireguard v0.0.0-20231211153847-12269c276173 // indirect
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260414002931-afd174a4e478 // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260414002931-afd174a4e478 // indirect
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260526163538-3dc84a4a5aaa // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa // indirect
|
||||
google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af // indirect
|
||||
gopkg.in/evanphx/json-patch.v4 v4.13.0 // indirect
|
||||
gopkg.in/inf.v0 v0.9.1 // indirect
|
||||
@@ -150,5 +150,5 @@ require (
|
||||
k8s.io/utils v0.0.0-20260210185600-b8788abfbbc2 // indirect
|
||||
sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 // indirect
|
||||
sigs.k8s.io/randfill v1.0.0 // indirect
|
||||
sigs.k8s.io/structured-merge-diff/v6 v6.3.2 // indirect
|
||||
sigs.k8s.io/structured-merge-diff/v6 v6.3.3 // indirect
|
||||
)
|
||||
|
||||
192
go.sum
192
go.sum
@@ -22,10 +22,14 @@ github.com/containerd/errdefs/pkg v0.3.0 h1:9IKJ06FvyNlexW690DXuQNx2KA2cUJXx151X
|
||||
github.com/containerd/errdefs/pkg v0.3.0/go.mod h1:NJw6s9HwNuRhnjJhM7pylWwMyAkmCQvQ4GpJHEqRLVk=
|
||||
github.com/containerd/log v0.1.0 h1:TCJt7ioM2cr/tfR8GPbGf9/VRAX8D2B4PjzCpfX540I=
|
||||
github.com/containerd/log v0.1.0/go.mod h1:VRRf09a7mHDIRezVKTRCrOq78v577GXq3bSa3EhrzVo=
|
||||
github.com/containernetworking/cni v1.3.0 h1:v6EpN8RznAZj9765HhXQrtXgX+ECGebEYEmnuFjskwo=
|
||||
github.com/containernetworking/cni v1.3.0/go.mod h1:Bs8glZjjFfGPHMw6hQu82RUgEPNGEaBb9KS5KtNMnJ4=
|
||||
github.com/containernetworking/plugins v1.9.1 h1:8oU6WsIsU3bpnNZuvHp74a6cE1MJwbj2P7s4/yTUNlA=
|
||||
github.com/containernetworking/plugins v1.9.1/go.mod h1:fj7kS55qg3o/RgS+WGsF3+ZxwIImMPusQZKzBpcSr4c=
|
||||
github.com/coreos/go-semver v0.3.1 h1:yi21YpKnrx1gt5R+la8n5WgS0kCrsPp33dmEyHReZr4=
|
||||
github.com/coreos/go-semver v0.3.1/go.mod h1:irMmmIw/7yzSRPWryHsK7EYSg09caPQL03VsM8rvUec=
|
||||
github.com/coreos/go-systemd/v22 v22.5.0 h1:RrqgGjYQKalulkV8NGVIfkXQf6YYmOyiJKk8iXXhfZs=
|
||||
github.com/coreos/go-systemd/v22 v22.5.0/go.mod h1:Y58oyj3AT4RCenI/lSvhwexgC+NSVTIJ3seZv2GcEnc=
|
||||
github.com/coreos/go-systemd/v22 v22.7.0 h1:LAEzFkke61DFROc7zNLX/WA2i5J8gYqe0rSj9KI28KA=
|
||||
github.com/coreos/go-systemd/v22 v22.7.0/go.mod h1:xNUYtjHu2EDXbsxz1i41wouACIwT7Ybq9o0BQhMwD0w=
|
||||
github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g=
|
||||
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
@@ -86,9 +90,6 @@ github.com/go-viper/mapstructure/v2 v2.4.0 h1:EBsztssimR/CONLSZZ04E8qAkxNYq4Qp9L
|
||||
github.com/go-viper/mapstructure/v2 v2.4.0/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM=
|
||||
github.com/goccy/go-yaml v1.18.0 h1:8W7wMFS12Pcas7KU+VVkaiCng+kG8QiFeFwzFb+rwuw=
|
||||
github.com/goccy/go-yaml v1.18.0/go.mod h1:XBurs7gK8ATbW4ZPGKgcbrY1Br56PdM69F7LkFRi1kA=
|
||||
github.com/godbus/dbus/v5 v5.0.4/go.mod h1:xhWf0FNVPg57R7Z0UbKHbJfkEywrmjJnf7w5xrFpKfA=
|
||||
github.com/gogo/protobuf v1.3.2 h1:Ov1cvc58UF3b5XjBnZv7+opcTcQFZebYjWzi34vdm4Q=
|
||||
github.com/gogo/protobuf v1.3.2/go.mod h1:P1XiOD3dCwIKUDQYPy72D8LYyHL2YPYrpS2s69NZV8Q=
|
||||
github.com/golang/protobuf v1.5.4 h1:i7eJL8qZTpSEXOPTxNKhASYpMn+8e5Q6AdndVa1dWek=
|
||||
github.com/golang/protobuf v1.5.4/go.mod h1:lnTiLA8Wa4RWRcIUkrtSVa5nRhsEGBg48fD6rSs7xps=
|
||||
github.com/google/gnostic-models v0.7.0 h1:qwTtogB15McXDaNqTZdzPJRHvaVJlAl+HVQnLmJEJxo=
|
||||
@@ -97,8 +98,8 @@ github.com/google/go-cmp v0.5.7/go.mod h1:n+brtR0CgQNWTVd5ZUFpTBC8YFBDLK/h/bpaJ8
|
||||
github.com/google/go-cmp v0.5.9/go.mod h1:17dUlkBOakJ0+DkrSSNjCkIjxS6bF9zb3elmeNGIjoY=
|
||||
github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8=
|
||||
github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU=
|
||||
github.com/google/go-containerregistry v0.21.7 h1:/vPFuVXDjtFREsVArW+0h1CIl5urnOhzei4X2DMW9IU=
|
||||
github.com/google/go-containerregistry v0.21.7/go.mod h1:kjSbt7/zMsKLWfnHrIvKvhXHUw91jbe9DNjPPJ32gXE=
|
||||
github.com/google/go-containerregistry v0.22.1 h1:RZuuSYhTvlDvtsK+NkutoCZ//C0X2ebLK8X8l3ULs84=
|
||||
github.com/google/go-containerregistry v0.22.1/go.mod h1:bJR35SK8XgisYmhg/FMQ/5RK0S/XrOAqLBV5/LR2XE0=
|
||||
github.com/google/gofuzz v1.0.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg=
|
||||
github.com/google/nftables v0.3.0 h1:bkyZ0cbpVeMHXOrtlFc8ISmfVqq5gPJukoYieyVmITg=
|
||||
github.com/google/nftables v0.3.0/go.mod h1:BCp9FsrbF1Fn/Yu6CLUc9GGZFw/+hsxfluNXXmxBfRM=
|
||||
@@ -120,16 +121,16 @@ github.com/gookit/rotatefile v0.3.0 h1:9MtCRBM79/Chcqp6ySHHmeDGpJE09WvyRFHo7yCQ+
|
||||
github.com/gookit/rotatefile v0.3.0/go.mod h1:MUaLyw2tEKNe8nta7o2qMfCGST30kzJqybG4KUreIu4=
|
||||
github.com/gookit/slog v0.7.1 h1:/q4YtsaJfdtzK+Q1g3QtNoO8cuEF6EjjU56W5U069i8=
|
||||
github.com/gookit/slog v0.7.1/go.mod h1:aJ4SGHlMR5YdfeQcICQBEn5bnF0Rpnuh+a5FEzQqXpE=
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.28.0 h1:HWRh5R2+9EifMyIHV7ZV+MIZqgz+PMpZ14Jynv3O2Zs=
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.28.0/go.mod h1:JfhWUomR1baixubs02l85lZYYOm7LV6om4ceouMv45c=
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.29.0 h1:5VipnvEpbqr2gA2VbM+nYVbkIF28c5ZQfqCBQ5g2xfk=
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.29.0/go.mod h1:Hyl3n6Twe1hvtd9XUXDec4pTvgMSEixRuQKPTMH2bNs=
|
||||
github.com/hugelgupf/socketpair v0.0.0-20190730060125-05d35a94e714 h1:/jC7qQFrv8CrSJVmaolDVOxTfS9kc36uB6H40kdbQq8=
|
||||
github.com/hugelgupf/socketpair v0.0.0-20190730060125-05d35a94e714/go.mod h1:2Goc3h8EklBH5mspfHFxBnEoURQCGzQQH1ga9Myjvis=
|
||||
github.com/huin/goupnp v1.3.0 h1:UvLUlWDNpoUdYzb2TCn+MuTWtcjXKSza2n6CBdQ0xXc=
|
||||
github.com/huin/goupnp v1.3.0/go.mod h1:gnGPsThkYa7bFi/KWmEysQRf48l2dvR5bxr2OFckNX8=
|
||||
github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8=
|
||||
github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw=
|
||||
github.com/insomniacslk/dhcp v0.0.0-20241224095048-b56fa0d5f25d h1:VkCNWh6tuQLgDBc6KrUOz/L1mCUQGnR1Ujj8uTgpwwk=
|
||||
github.com/insomniacslk/dhcp v0.0.0-20241224095048-b56fa0d5f25d/go.mod h1:VvGYjkZoJyKqlmT1yzakUs4mfKMNB0XdODP0+rdml6k=
|
||||
github.com/insomniacslk/dhcp v0.0.0-20260719225207-c76316d4aa82 h1:y5aU8Uvl7eyM5WNgdQvRxbMJb+zo7pD+S72/Yo4pvnQ=
|
||||
github.com/insomniacslk/dhcp v0.0.0-20260719225207-c76316d4aa82/go.mod h1:qfvBmyDNp+/liLEYWRvqny/PEz9hGe2Dz833eXILSmo=
|
||||
github.com/josharian/intern v1.0.0 h1:vlS4z54oSdjm0bgjRigI+G1HpF+tI+9rE5LLzOg8HmY=
|
||||
github.com/josharian/intern v1.0.0/go.mod h1:5DoeVV0s6jJacbCEi61lwdGj/aVlrQvzHFFd8Hwg//Y=
|
||||
github.com/josharian/native v1.0.0/go.mod h1:7X/raswPFr05uY3HiLlYeyQntB6OO7E/d2Cu7qoaN2w=
|
||||
@@ -143,10 +144,8 @@ github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnr
|
||||
github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo=
|
||||
github.com/k-sone/critbitgo v1.4.0 h1:l71cTyBGeh6X5ATh6Fibgw3+rtNT80BA0uNNWgkPrbE=
|
||||
github.com/k-sone/critbitgo v1.4.0/go.mod h1:7E6pyoyADnFxlUBEKcnfS49b7SUAQGMK+OAp/UQvo0s=
|
||||
github.com/kisielk/errcheck v1.5.0/go.mod h1:pFxgyoBC7bSaBwPgfKdkLd5X25qrDl4LWUI2bnpBCr8=
|
||||
github.com/kisielk/gotool v1.0.0/go.mod h1:XhKaO+MFFWcvkIS/tQcRk01m1F5IRFswLeQ+oQHNcck=
|
||||
github.com/klauspost/compress v1.18.6 h1:2jupLlAwFm95+YDR+NwD2MEfFO9d4z4Prjl1XXDjuao=
|
||||
github.com/klauspost/compress v1.18.6/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
|
||||
github.com/klauspost/compress v1.19.2 h1:hMRETovs/pu/dVWN7zIT1PGG8t509MwT6bO7XSi26R8=
|
||||
github.com/klauspost/compress v1.19.2/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
|
||||
github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE=
|
||||
github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk=
|
||||
github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY=
|
||||
@@ -193,18 +192,18 @@ github.com/morikuni/aec v1.1.0 h1:vBBl0pUnvi/Je71dsRrhMBtreIqNMYErSAbEeb8jrXQ=
|
||||
github.com/morikuni/aec v1.1.0/go.mod h1:xDRgiq/iw5l+zkao76YTKzKttOp2cwPEne25HDkJnBw=
|
||||
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA=
|
||||
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ=
|
||||
github.com/onsi/ginkgo/v2 v2.32.0 h1:Hw7s2pVrQo/8Yz5N77qdnpHaoc+c6cC9WIV1Jce+J6E=
|
||||
github.com/onsi/ginkgo/v2 v2.32.0/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44=
|
||||
github.com/onsi/gomega v1.42.1 h1:iN1rCUX+44NZ1Dc97MPoeFYbFR0vh8zxoxMFwKdyZ6I=
|
||||
github.com/onsi/gomega v1.42.1/go.mod h1:REff/hsDsodHoKlWsP2mAPhu1+5/6hVYNf9rIEBpeSg=
|
||||
github.com/onsi/ginkgo/v2 v2.32.2 h1:2o6vyFvR6snrJWgRVztC+OwuqqPEMI1UzYl2s2iU7Cg=
|
||||
github.com/onsi/ginkgo/v2 v2.32.2/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44=
|
||||
github.com/onsi/gomega v1.43.0 h1:VlG/1FxqNxhSO+lq/OHBNaaqwiBK/mO8JbVkX9Y+FeU=
|
||||
github.com/onsi/gomega v1.43.0/go.mod h1:REff/hsDsodHoKlWsP2mAPhu1+5/6hVYNf9rIEBpeSg=
|
||||
github.com/opencontainers/go-digest v1.0.0 h1:apOUWs51W5PlhuyGyz9FCeeBIOUDA/6nW8Oi/yOhh5U=
|
||||
github.com/opencontainers/go-digest v1.0.0/go.mod h1:0JzlMkj0TRzQZfJkVvzbP0HBR3IKzErnv2BNG4W4MAM=
|
||||
github.com/opencontainers/image-spec v1.1.1 h1:y0fUlFfIZhPF1W537XOLg0/fcx6zcHCJwooC2xJA040=
|
||||
github.com/opencontainers/image-spec v1.1.1/go.mod h1:qpqAh3Dmcf36wStyyWU+kCeDgrGnAve2nCC8+7h8Q0M=
|
||||
github.com/orcaman/concurrent-map/v2 v2.0.1 h1:jOJ5Pg2w1oeB6PeDurIYf6k9PQ+aTITr/6lP/L/zp6c=
|
||||
github.com/orcaman/concurrent-map/v2 v2.0.1/go.mod h1:9Eq3TG2oBe5FirmYWQfYO5iH1q0Jv47PLaNK++uCdOM=
|
||||
github.com/osrg/gobgp/v4 v4.7.0 h1:KPnmJVDFx1AJ39ESNpzHukAzaQYa4kRR5c7BbuDJhuM=
|
||||
github.com/osrg/gobgp/v4 v4.7.0/go.mod h1:j1GLEuE20jm2YAoGmaHGb3y9lGH/KBgCBT4Ss5RY/wQ=
|
||||
github.com/osrg/gobgp/v4 v4.9.0 h1:pKOw914kwQ4I/lWNVTfEDosEN3FuqPGytMEInXxpTyQ=
|
||||
github.com/osrg/gobgp/v4 v4.9.0/go.mod h1:bJbFm7T2nRANggShfl3I9h0UpPCzu4uAY5J/6dTdRvs=
|
||||
github.com/pelletier/go-toml v1.9.5 h1:4yBQzkHv+7BHq2PQUZF3Mx0IYxG7LsP222s7Agd3ve8=
|
||||
github.com/pelletier/go-toml v1.9.5/go.mod h1:u1nR/EPcESfeI/szUZKdtJ0xRNbUoANCkoOuaOx1Y+c=
|
||||
github.com/pelletier/go-toml/v2 v2.2.3 h1:YmeHyLY8mFWbdkNWwpr+qIL2bEqT0o95WSdkNHvL12M=
|
||||
@@ -216,14 +215,14 @@ github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINE
|
||||
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 h1:Jamvg5psRIccs7FGNTlIRMkT8wgtp5eCXdBlqhYGL6U=
|
||||
github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||
github.com/prometheus/client_golang v1.23.2 h1:Je96obch5RDVy3FDMndoUsjAhG5Edi49h0RJWRi/o0o=
|
||||
github.com/prometheus/client_golang v1.23.2/go.mod h1:Tb1a6LWHB3/SPIzCoaDXI4I8UHKeFTEQ1YCr+0Gyqmg=
|
||||
github.com/prometheus/client_golang v1.24.1 h1:JnJkREXzWxUdCuPFpIWZiPispT9xVV59uiuyR2bPlnU=
|
||||
github.com/prometheus/client_golang v1.24.1/go.mod h1:F+oSRECHg4sse5ucfYpYDeIv/hu68Zo0uoHKetWnzcE=
|
||||
github.com/prometheus/client_model v0.6.2 h1:oBsgwpGs7iVziMvrGhE53c/GrLUsZdHnqNwqPLxwZyk=
|
||||
github.com/prometheus/client_model v0.6.2/go.mod h1:y3m2F6Gdpfy6Ut/GBsUqTWZqCUvMVzSfMLjcu6wAwpE=
|
||||
github.com/prometheus/common v0.66.1 h1:h5E0h5/Y8niHc5DlaLlWLArTQI7tMrsfQjHV+d9ZoGs=
|
||||
github.com/prometheus/common v0.66.1/go.mod h1:gcaUsgf3KfRSwHY4dIMXLPV0K/Wg1oZ8+SbZk/HH/dA=
|
||||
github.com/prometheus/procfs v0.16.1 h1:hZ15bTNuirocR6u0JZ6BAHHmwS1p8B4P6MRqxtzMyRg=
|
||||
github.com/prometheus/procfs v0.16.1/go.mod h1:teAbpZRB1iIAJYREa1LsoWUXykVXA1KlTmWl8x/U+Is=
|
||||
github.com/prometheus/common v0.70.1 h1:1HvjP4D5oL3t8RsPlwxA9onvvStjtIHYE5XuuwOi/PY=
|
||||
github.com/prometheus/common v0.70.1/go.mod h1:VdFUQDMZK3VLkurFUVhia6uys/0suUp86TJz5qbJRhc=
|
||||
github.com/prometheus/procfs v0.21.1 h1:GljZCt+zSTS+NZq88cyQ1LjZ+RCHp3uVuabBWA5+OJI=
|
||||
github.com/prometheus/procfs v0.21.1/go.mod h1:aB55Cww9pdSJVHk0hUf0inxWyyjPogFIjmHKYgMKmtY=
|
||||
github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ=
|
||||
github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc=
|
||||
github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM=
|
||||
@@ -247,11 +246,11 @@ github.com/spf13/pflag v1.0.10/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3A
|
||||
github.com/spf13/viper v1.20.1 h1:ZMi+z/lvLyPSCoNtFCpqjy0S4kPbirhpTMwl8BkW9X4=
|
||||
github.com/spf13/viper v1.20.1/go.mod h1:P9Mdzt1zoHIG8m2eZQinpiBjo6kCmZSKBClNNqjJvu4=
|
||||
github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME=
|
||||
github.com/stretchr/objx v0.5.2 h1:xuMeJ0Sdp5ZMRXx/aWO6RZxdr3beISkG5/G/aIRr3pY=
|
||||
github.com/stretchr/objx v0.5.2/go.mod h1:FRsXN1f5AsAjCGJKqEizvkpNtU+EGNCLh3NxZ/8L+MA=
|
||||
github.com/stretchr/objx v0.5.3 h1:jmXUvGomnU1o3W/V5h2VEradbpJDwGrzugQQvL0POH4=
|
||||
github.com/stretchr/objx v0.5.3/go.mod h1:rDQraq+vQZU7Fde9LOZLr8Tax6zZvy4kuNKF+QYS+U0=
|
||||
github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI=
|
||||
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
|
||||
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
|
||||
github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE=
|
||||
github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg=
|
||||
github.com/subosito/gotenv v1.6.0 h1:9NlTDc1FTs4qu0DDq7AEtTPNw6SVm7uBMsUCUjABIf8=
|
||||
github.com/subosito/gotenv v1.6.0/go.mod h1:Dk4QP5c2W3ibzajGcXpNraDfq2IrhjMIvMSWPKKo0FU=
|
||||
github.com/tidwall/gjson v1.18.0 h1:FIDeeyB800efLX89e5a8Y0BNH+LOngJyGrIWxG2FKQY=
|
||||
@@ -266,40 +265,38 @@ github.com/u-root/uio v0.0.0-20240224005618-d2acac8f3701 h1:pyC9PaHYZFgEKFdlp3G8
|
||||
github.com/u-root/uio v0.0.0-20240224005618-d2acac8f3701/go.mod h1:P3a5rG4X7tI17Nn3aOIAYr5HbIMukwXG0urG0WuL8OA=
|
||||
github.com/valyala/bytebufferpool v1.0.0 h1:GqA5TC/0021Y/b9FG4Oi9Mr3q7XYx6KllzawFIhcdPw=
|
||||
github.com/valyala/bytebufferpool v1.0.0/go.mod h1:6bBcMArwyJ5K/AmCkWv1jt77kVWyCJ6HpOuEn7z0Csc=
|
||||
github.com/vishvananda/netlink v1.3.1 h1:3AEMt62VKqz90r0tmNhog0r/PpWKmrEShJU0wJW6bV0=
|
||||
github.com/vishvananda/netlink v1.3.1/go.mod h1:ARtKouGSTGchR8aMwmkzC0qiNPrrWO5JS/XMVl45+b4=
|
||||
github.com/vishvananda/netlink v1.3.2-0.20260830232854-cf01b55a4a4b h1:XtEhFJO3IqjQWHJZ3bbNm7LtbDehriJK65KW+6lnw+Q=
|
||||
github.com/vishvananda/netlink v1.3.2-0.20260830232854-cf01b55a4a4b/go.mod h1:lEui7SPMd9fgxzHVGRAvTxsBGCF6PRH81o2kLWLWHgw=
|
||||
github.com/vishvananda/netns v0.0.5 h1:DfiHV+j8bA32MFM7bfEunvT8IAqQ/NzSJHtcmW5zdEY=
|
||||
github.com/vishvananda/netns v0.0.5/go.mod h1:SpkAiCQRtJ6TvvxPnOSyH3BMl6unz3xZlaprSwhNNJM=
|
||||
github.com/x448/float16 v0.8.4 h1:qLwI1I70+NjRFUR3zs1JPUCgaCXSh3SW62uAKT1mSBM=
|
||||
github.com/x448/float16 v0.8.4/go.mod h1:14CWIYCyZA/cWjXOioeEpHeN/83MdbZDRQHoFcYsOfg=
|
||||
github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e h1:JVG44RsyaB9T2KIHavMF/ppJZNG9ZpyihvCd0w101no=
|
||||
github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e/go.mod h1:RbqR21r5mrJuqunuUZ/Dhy/avygyECGrLceyNeo4LiM=
|
||||
github.com/yuin/goldmark v1.1.27/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74=
|
||||
github.com/yuin/goldmark v1.2.1/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74=
|
||||
go.etcd.io/etcd/api/v3 v3.6.12 h1:OLOZUKEuAA36TR48F0cIaa8FdzrWygjyfrJxXg4iDgs=
|
||||
go.etcd.io/etcd/api/v3 v3.6.12/go.mod h1:p14EIQXHbuOQbVvL/WEes5uqKnxP9AgKJgpjbMVvzvE=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.13 h1:7QeMOisYByx8dBA7/CKcwCaPWfjb5C0xpmrIov/8WyY=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.13/go.mod h1:Dn2zUBOCu/6xYcd6iAjB7LgoY16OTQjDZfWHLwvuQj4=
|
||||
go.etcd.io/etcd/client/v3 v3.6.12 h1:kMSP6JcPZMqSJiX+TXdUIBU/4eXEZWBAaui4VihMbIc=
|
||||
go.etcd.io/etcd/client/v3 v3.6.12/go.mod h1:CMs6fJWYiZQk4ytFjd4lE1diOvvRMmtbbn/alZXd3dQ=
|
||||
go.etcd.io/etcd/api/v3 v3.7.1 h1:KJG0/DcWGfe3Y1otDf/fsBf0TSSgpxZ5RO/L8SFt73E=
|
||||
go.etcd.io/etcd/api/v3 v3.7.1/go.mod h1:8bXIpCMeV7E3/XL0Ix123ATn3dB+0V7d9zklHbB0m78=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.7.1 h1:rKYsj3pRkR0eK3yjT3XOgrhqfmIfj9pzNgxjh7mfFv4=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.7.1/go.mod h1:cnzZGIUzSfjEwLC6UBVsSXlEK1eepS/JUD7wE6PLRT0=
|
||||
go.etcd.io/etcd/client/v3 v3.7.1 h1:0PEMMC0KuZmVIN+RAbdqfkZ45pYTgKVtmBEbRCvZFUg=
|
||||
go.etcd.io/etcd/client/v3 v3.7.1/go.mod h1:ffNqALa8tRCYhYo1F9oR489y23K39Gz+BSR3ApAGYq0=
|
||||
go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64=
|
||||
go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.61.0 h1:F7Jx+6hwnZ41NSFTO5q4LYDtJRXBf2PD0rNBkeB/lus=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.61.0/go.mod h1:UHB22Z8QsdRDrnAtX4PntOl36ajSxcdUMt1sF7Y6E7Q=
|
||||
go.opentelemetry.io/otel v1.43.0 h1:mYIM03dnh5zfN7HautFE4ieIig9amkNANT+xcVxAj9I=
|
||||
go.opentelemetry.io/otel v1.43.0/go.mod h1:JuG+u74mvjvcm8vj8pI5XiHy1zDeoCS2LB1spIq7Ay0=
|
||||
go.opentelemetry.io/otel v1.44.0 h1:JjwHmHpA4iZ3wBxluu2fbbE7j4kqlE8jXyAyPXH7HqU=
|
||||
go.opentelemetry.io/otel v1.44.0/go.mod h1:BMgjTHL9WPRlRjL2oZCBTL4whCGtXch2H4BhOPIAyYc=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.43.0 h1:88Y4s2C8oTui1LGM6bTWkw0ICGcOLCAI5l6zsD1j20k=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.43.0/go.mod h1:Vl1/iaggsuRlrHf/hfPJPvVag77kKyvrLeD10kpMl+A=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.43.0 h1:3iZJKlCZufyRzPzlQhUIWVmfltrXuGyfjREgGP3UUjc=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.43.0/go.mod h1:/G+nUPfhq2e+qiXMGxMwumDrP5jtzU+mWN7/sjT2rak=
|
||||
go.opentelemetry.io/otel/metric v1.43.0 h1:d7638QeInOnuwOONPp4JAOGfbCEpYb+K6DVWvdxGzgM=
|
||||
go.opentelemetry.io/otel/metric v1.43.0/go.mod h1:RDnPtIxvqlgO8GRW18W6Z/4P462ldprJtfxHxyKd2PY=
|
||||
go.opentelemetry.io/otel/sdk v1.43.0 h1:pi5mE86i5rTeLXqoF/hhiBtUNcrAGHLKQdhg4h4V9Dg=
|
||||
go.opentelemetry.io/otel/sdk v1.43.0/go.mod h1:P+IkVU3iWukmiit/Yf9AWvpyRDlUeBaRg6Y+C58QHzg=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.43.0 h1:S88dyqXjJkuBNLeMcVPRFXpRw2fuwdvfCGLEo89fDkw=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.43.0/go.mod h1:C/RJtwSEJ5hzTiUz5pXF1kILHStzb9zFlIEe85bhj6A=
|
||||
go.opentelemetry.io/otel/trace v1.43.0 h1:BkNrHpup+4k4w+ZZ86CZoHHEkohws8AY+WTX09nk+3A=
|
||||
go.opentelemetry.io/otel/trace v1.43.0/go.mod h1:/QJhyVBUUswCphDVxq+8mld+AvhXZLhe+8WVFxiFff0=
|
||||
go.opentelemetry.io/otel/metric v1.44.0 h1:1w0gILTcHdr3YI+ixLyjemwrVnsMURbTZFrSYCdDdmc=
|
||||
go.opentelemetry.io/otel/metric v1.44.0/go.mod h1:8O7hanEPBNgEMmybD3s2VBKcgWOCsA6tzHBPODAiquo=
|
||||
go.opentelemetry.io/otel/sdk v1.44.0 h1:nHYwb9lK+fJPU/dnT6s7W7Z8itMWyqrnVfbheVYrZ58=
|
||||
go.opentelemetry.io/otel/sdk v1.44.0/go.mod h1:Osuydd3Se74nqjAKxid74N5eC+jfEqfTegHRnq58oK0=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.44.0 h1:3LlKgI+VjbVsjNRFZJZAJ30WjXC5VkNRks6si09iEfI=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.44.0/go.mod h1:5B5pMARnXxKhltooO4xUuCBorl65a4EpnTalObqOigA=
|
||||
go.opentelemetry.io/otel/trace v1.44.0 h1:jxF5CsGYCe74MCRx2X4g7WsY/VBKRqqpNvXlX/6gtIk=
|
||||
go.opentelemetry.io/otel/trace v1.44.0/go.mod h1:oLl1jrMQAVo6v3GAggN+1VH9VIz9iUSvW53sW1Q8PIE=
|
||||
go.opentelemetry.io/proto/otlp v1.10.0 h1:IQRWgT5srOCYfiWnpqUYz9CVmbO8bFmKcwYxpuCSL2g=
|
||||
go.opentelemetry.io/proto/otlp v1.10.0/go.mod h1:/CV4QoCR/S9yaPj8utp3lvQPoqMtxXdzn7ozvvozVqk=
|
||||
go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto=
|
||||
@@ -308,81 +305,60 @@ go.uber.org/multierr v1.11.0 h1:blXXJkSxSSfBVBlC76pxqeO+LN3aDfLQo+309xJstO0=
|
||||
go.uber.org/multierr v1.11.0/go.mod h1:20+QtiLqy0Nd6FdQB9TLXag12DsQkrbs3htMFfDN80Y=
|
||||
go.uber.org/zap v1.28.0 h1:IZzaP1Fv73/T/pBMLk4VutPl36uNC+OSUh3JLG3FIjo=
|
||||
go.uber.org/zap v1.28.0/go.mod h1:rDLpOi171uODNm/mxFcuYWxDsqWSAVkFdX4XojSKg/Q=
|
||||
go.yaml.in/yaml/v2 v2.4.3 h1:6gvOSjQoTB3vt1l+CU+tSyi/HOjfOjRLJ4YwYZGwRO0=
|
||||
go.yaml.in/yaml/v2 v2.4.3/go.mod h1:zSxWcmIDjOzPXpjlTTbAsKokqkDNAVtZO0WOMiT90s8=
|
||||
go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc=
|
||||
go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ=
|
||||
go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ=
|
||||
go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
|
||||
go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw=
|
||||
go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg=
|
||||
golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w=
|
||||
golang.org/x/crypto v0.0.0-20191011191535-87dc89f01550/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI=
|
||||
golang.org/x/crypto v0.0.0-20200622213623-75b288015ac9/go.mod h1:LzIPMQfyMNhhGPhUkYOs5KpL4U8rLKemX1yGLhDgUto=
|
||||
golang.org/x/crypto v0.53.0 h1:QZ4Muo8THX6CizN2vPPd5fBGHyogrdK9fG4wLPFUsto=
|
||||
golang.org/x/crypto v0.53.0/go.mod h1:DNLU434OwVakk9PzuwV8w62mAJpRJL3vsgcfp4Qnsio=
|
||||
golang.org/x/crypto v0.55.0 h1:+KWHjbgOaAQ66dh/YlkZKHlz9ZUlq61AFirAR9ntP8M=
|
||||
golang.org/x/crypto v0.55.0/go.mod h1:uq0V9dE/fzQuJtbnL+2EhWOE63vo164FY8xqEnV9xis=
|
||||
golang.org/x/exp v0.0.0-20250103183323-7d7fa50e5329 h1:9kj3STMvgqy3YA4VQXBrN7925ICMxD5wzMRcgA30588=
|
||||
golang.org/x/exp v0.0.0-20250103183323-7d7fa50e5329/go.mod h1:qj5a5QZpwLU2NLQudwIN5koi3beDhSAlJwa67PuM98c=
|
||||
golang.org/x/mod v0.2.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA=
|
||||
golang.org/x/mod v0.3.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA=
|
||||
golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ=
|
||||
golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0=
|
||||
golang.org/x/net v0.0.0-20190404232315-eb5bcb51f2a3/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg=
|
||||
golang.org/x/mod v0.39.0 h1:UF5zwQdCRRUpHfyPwr7d4UrGiVeldIsogtzWVnczL74=
|
||||
golang.org/x/mod v0.39.0/go.mod h1:bvIbwjQ0HUFFf5AKukeeYQG4ZBUG9yxQbR9aEweIwYY=
|
||||
golang.org/x/net v0.0.0-20190503192946-f4e77d36d62c/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg=
|
||||
golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s=
|
||||
golang.org/x/net v0.0.0-20200226121028-0de0cce0169b/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s=
|
||||
golang.org/x/net v0.0.0-20201021035429-f5854403a974/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU=
|
||||
golang.org/x/net v0.0.0-20220923203811-8be639271d50/go.mod h1:YDH+HFinaLZZlnHAfSS6ZXJJ9M9t4Dl22yv3iI2vPwk=
|
||||
golang.org/x/net v0.56.0 h1:Rw8j/hFzGvJUZwNBXnAtf5sVDVt+65SK2C7IxCxZt5o=
|
||||
golang.org/x/net v0.56.0/go.mod h1:D3Ku6r+V6JROoZK144D2XfMHFcMq/0zSfLelVTCFKec=
|
||||
golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To=
|
||||
golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU=
|
||||
golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs=
|
||||
golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q=
|
||||
golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.0.0-20190911185100-cd5d95a43a6e/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.0.0-20201020160332-67f06af15bc9/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.0.0-20210220032951-036812b2e83c/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.0.0-20220923202941-7f9b1623fab7/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
|
||||
golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk=
|
||||
golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0=
|
||||
golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||
golang.org/x/sys v0.0.0-20190412213103-97732733099d/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200930185726-fdedc70b468f/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20220319134239-a9b59b0215f8/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20220728004956-3c1f35247d10/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.2.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.10.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo=
|
||||
golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og=
|
||||
golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8=
|
||||
golang.org/x/term v0.44.0 h1:0rLvDRCtNj0gZkyIXhCyOb2OAzEhLVqc4B+hrsBhrmc=
|
||||
golang.org/x/term v0.44.0/go.mod h1:7ze4MdzUzLXpSAoFP1H0bOI9aXDqveSvatT5vKcFh2Y=
|
||||
golang.org/x/term v0.45.0 h1:NwWyBmoJCbfTHpxrWoZ9C6/VxOf7ic219I8xZZFdrf0=
|
||||
golang.org/x/term v0.45.0/go.mod h1:9aqxs0blBcrm/n0L9QW0aRVD+ktan8ssZromtqJC43w=
|
||||
golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
|
||||
golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
|
||||
golang.org/x/text v0.3.7/go.mod h1:u+2+/6zg+i71rQMx5EYifcz6MCKuco9NR6JIITiCfzQ=
|
||||
golang.org/x/text v0.38.0 h1:sXmwo9DwP3OK9EZ7PqAdaooSGozfl/3a6/xJcbzPRhE=
|
||||
golang.org/x/text v0.38.0/go.mod h1:YXZt3QhHUKYT53r2lLKFIVi6Ao1jdzrTR/KQ09qyxF4=
|
||||
golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8=
|
||||
golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M=
|
||||
golang.org/x/time v0.14.0 h1:MRx4UaLrDotUKUdCIqzPC48t1Y9hANFKIRpNx+Te8PI=
|
||||
golang.org/x/time v0.14.0/go.mod h1:eL/Oa2bBBK0TkX57Fyni+NgnyQQN4LitPmob2Hjnqw4=
|
||||
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
|
||||
golang.org/x/tools v0.0.0-20191119224855-298f0cb1881e/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo=
|
||||
golang.org/x/tools v0.0.0-20200619180055-7c47624df98f/go.mod h1:EkVYQZoAsY45+roYkvgYkIh4xh/qjgUK9TdY2XT94GE=
|
||||
golang.org/x/tools v0.0.0-20210106214847-113979e3529a/go.mod h1:emZCQorbCU4vsT4fOWvOPXz4eW1wZW4PmDk9uLelYpA=
|
||||
golang.org/x/tools v0.46.0 h1:7jTurBkPZu4moS/Uy4OQT1M+QBlsj3wejyZwsT8Z7rk=
|
||||
golang.org/x/tools v0.46.0/go.mod h1:FrD85F8l+NWL+9XWBSyVSHO6Ne4jutsfIFba7AWQ5Ys=
|
||||
golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||
golang.org/x/xerrors v0.0.0-20191011141410-1b5146add898/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||
golang.org/x/tools v0.49.0 h1:3NI7VXzL9+1WZD52Dx2ttoPwD5DWrFGpl9mFZDlmisI=
|
||||
golang.org/x/tools v0.49.0/go.mod h1:SJNXV9DBKT0UbdttsQjbfJlAE/q+y36++zo3uL3N0Oo=
|
||||
golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||
golang.org/x/xerrors v0.0.0-20200804184101-5ec99f83aff1/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||
golang.zx2c4.com/wireguard v0.0.0-20231211153847-12269c276173 h1:/jFs0duh4rdb8uIfPMv78iAJGcPKDeqAFnaLBropIC4=
|
||||
golang.zx2c4.com/wireguard v0.0.0-20231211153847-12269c276173/go.mod h1:tkCQ4FQXmpAgYVh++1cq16/dH4QJtmvpRv19DWGAHSA=
|
||||
golang.zx2c4.com/wireguard/wgctrl v0.0.0-20241231184526-a9ab2273dd10 h1:3GDAcqdIg1ozBNLgPy4SLT84nfcBjr6rhGtXYtrkWLU=
|
||||
golang.zx2c4.com/wireguard/wgctrl v0.0.0-20241231184526-a9ab2273dd10/go.mod h1:T97yPqesLiNrOYxkwmhMI0ZIlJDm+p0PMR8eRVeR5tQ=
|
||||
gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4=
|
||||
gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260414002931-afd174a4e478 h1:yQugLulqltosq0B/f8l4w9VryjV+N/5gcW0jQ3N8Qec=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260414002931-afd174a4e478/go.mod h1:C6ADNqOxbgdUUeRTU+LCHDPB9ttAMCTff6auwCVa4uc=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260414002931-afd174a4e478 h1:RmoJA1ujG+/lRGNfUnOMfhCy5EipVMyvUE+KNbPbTlw=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260414002931-afd174a4e478/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8=
|
||||
google.golang.org/grpc v1.82.1 h1:NnAxzGRA0677vCa4BUkOAnO5+FfQqVl9iUXeD0IqcGE=
|
||||
google.golang.org/grpc v1.82.1/go.mod h1:yzTZ1TB1Z3SG+LIYaI+WiE8D5+PZ3ArnrSp8zF3+/ZA=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260526163538-3dc84a4a5aaa h1:Kjn0N0tCrDgiAFW+lGO4JZ3ck44CehvJQMAwj9QF0G8=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260526163538-3dc84a4a5aaa/go.mod h1:q4lMZS6kskjT5HvCPrnnypcDPVJqT/f4nfxmkE7gryY=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa h1:mZHHdPZl0dbGHCflZgAq/Q468DWVFcU2whhB2KAo8fk=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8=
|
||||
google.golang.org/grpc v1.83.2 h1:EManeRomTObA0BU7I8vXgg/78uE5MJ9M8B39EX2WscU=
|
||||
google.golang.org/grpc v1.83.2/go.mod h1:YPI1hK3kDked6iHvgX3tR0y+nX/qpMFKhPgFsokw1S8=
|
||||
google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af h1:+5/Sw3GsDNlEmu7TfklWKPdQ0Ykja5VEmq2i817+jbI=
|
||||
google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||
@@ -396,12 +372,12 @@ gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
||||
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||
gotest.tools/v3 v3.4.0 h1:ZazjZUfuVeZGLAmlKKuyv3IKP5orXcwtOwDQH6YVr6o=
|
||||
gotest.tools/v3 v3.4.0/go.mod h1:CtbdzLSsqVhDgMtKsx03ird5YTGB3ar27v0u/yKBW5g=
|
||||
k8s.io/api v0.36.2 h1:TF6YDLIzKfccK7cq9YpTcGX8TJmEkHVRv78DM51fRYY=
|
||||
k8s.io/api v0.36.2/go.mod h1:F4LbMO4brjZYh7yFkXWhynSvtB7YauxV4c+HHkNRGNg=
|
||||
k8s.io/apimachinery v0.36.2 h1:0PE/W/WNy1UX61NLbXY5TMbJ6UwLL6E6lAPkYrKFxbQ=
|
||||
k8s.io/apimachinery v0.36.2/go.mod h1:fvf/HOLXq9RId0rnDIbN1OEBvHXdQbLMM8nu0LcBUf4=
|
||||
k8s.io/client-go v0.36.2 h1:bfgxmFKc9CgqsgX4xKLAAdmTQlWee7Ob/HlDOrJ5TBI=
|
||||
k8s.io/client-go v0.36.2/go.mod h1:1vgO4OAlfPnoLcb+Rze2GF5rAr14w8qjrYMoyXJzQj0=
|
||||
k8s.io/api v0.36.4 h1:RxrvqCL6vgH5/+UnTeu1IIFqYmGfy0hnyrod1rn35Oo=
|
||||
k8s.io/api v0.36.4/go.mod h1:S2B3orCFBDhrgyWbLeuKcT2QdHIpQesBkCYSlWtwUOw=
|
||||
k8s.io/apimachinery v0.36.4 h1:PT2UzkupGuAx/+xT5XjiMJ1WGpY3fn9/hdAvjweRet4=
|
||||
k8s.io/apimachinery v0.36.4/go.mod h1:p2I2dipt7JHG+quVwQ1d02d28O4GdDi77RByQ13MTpk=
|
||||
k8s.io/client-go v0.36.4 h1:MDvfDNvMSt0Br94SK8neviVlwL9qifw9B26hJCpD1K0=
|
||||
k8s.io/client-go v0.36.4/go.mod h1:pNK4WKELbwlEDvtbE8l22lEZL5THYF61H5EealokZmA=
|
||||
k8s.io/klog/v2 v2.140.0 h1:Tf+J3AH7xnUzZyVVXhTgGhEKnFqye14aadWv7bzXdzc=
|
||||
k8s.io/klog/v2 v2.140.0/go.mod h1:o+/RWfJ6PwpnFn7OyAG3QnO47BFsymfEfrz6XyYSSp0=
|
||||
k8s.io/kube-openapi v0.0.0-20260317180543-43fb72c5454a h1:xCeOEAOoGYl2jnJoHkC3hkbPJgdATINPMAxaynU2Ovg=
|
||||
@@ -412,11 +388,11 @@ pgregory.net/rapid v1.1.0 h1:CMa0sjHSru3puNx+J0MIAuiiEV4N0qj8/cMWGBBCsjw=
|
||||
pgregory.net/rapid v1.1.0/go.mod h1:PY5XlDGj0+V1FCq0o192FdRhpKHGTRIWBgqjDBTrq04=
|
||||
sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 h1:IpInykpT6ceI+QxKBbEflcR5EXP7sU1kvOlxwZh5txg=
|
||||
sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730/go.mod h1:mdzfpAEoE6DHQEN0uh9ZbOCuHbLK5wOm7dK4ctXE9Tg=
|
||||
sigs.k8s.io/kind v0.32.0 h1:p9hscbj98u/qyrjVpjId86LI70nQmbSsipV7wCG10Xk=
|
||||
sigs.k8s.io/kind v0.32.0/go.mod h1:FSqriGaoTPruiXWfRnUXNykF8r2t+fHtK0P0m1AbGF8=
|
||||
sigs.k8s.io/kind v0.33.0 h1:AjvDv3vOygb/VKLVQW87lfktIBzkxR8Ump9DjxC8+Lk=
|
||||
sigs.k8s.io/kind v0.33.0/go.mod h1:FSqriGaoTPruiXWfRnUXNykF8r2t+fHtK0P0m1AbGF8=
|
||||
sigs.k8s.io/randfill v1.0.0 h1:JfjMILfT8A6RbawdsK2JXGBR5AQVfd+9TbzrlneTyrU=
|
||||
sigs.k8s.io/randfill v1.0.0/go.mod h1:XeLlZ/jmk4i1HRopwe7/aU3H5n1zNUcX6TM94b3QxOY=
|
||||
sigs.k8s.io/structured-merge-diff/v6 v6.3.2 h1:kwVWMx5yS1CrnFWA/2QHyRVJ8jM6dBA80uLmm0wJkk8=
|
||||
sigs.k8s.io/structured-merge-diff/v6 v6.3.2/go.mod h1:M3W8sfWvn2HhQDIbGWj3S099YozAsymCo/wrT5ohRUE=
|
||||
sigs.k8s.io/structured-merge-diff/v6 v6.3.3 h1:u08YRbVUi59ri4YD6cg0UqNM4Dimn0sIl+wldcx5PYw=
|
||||
sigs.k8s.io/structured-merge-diff/v6 v6.3.3/go.mod h1:M3W8sfWvn2HhQDIbGWj3S099YozAsymCo/wrT5ohRUE=
|
||||
sigs.k8s.io/yaml v1.6.0 h1:G8fkbMSAFqgEFgh4b1wmtzDnioxFCUgTZhlbj5P9QYs=
|
||||
sigs.k8s.io/yaml v1.6.0/go.mod h1:796bPqUfzR/0jLAl6XjHl3Ck7MiyVv8dbTdyT3/pMf4=
|
||||
|
||||
4
main.go
4
main.go
@@ -1,6 +1,8 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
|
||||
"github.com/kube-vip/kube-vip/cmd"
|
||||
)
|
||||
|
||||
@@ -14,5 +16,5 @@ func main() {
|
||||
|
||||
cmd.Release.Version = Version
|
||||
cmd.Release.Build = Build
|
||||
cmd.Execute()
|
||||
os.Exit(cmd.Execute())
|
||||
}
|
||||
|
||||
114
pkg/arp/arp.go
114
pkg/arp/arp.go
@@ -13,15 +13,17 @@ import (
|
||||
"github.com/vishvananda/netlink"
|
||||
)
|
||||
|
||||
const linkSubscriptionBuffer = 64
|
||||
|
||||
type Manager struct {
|
||||
instances sync.Map
|
||||
mu sync.Mutex
|
||||
instances map[string]*Instance
|
||||
config *kubevip.Config
|
||||
}
|
||||
|
||||
type Instance struct {
|
||||
network vip.Network
|
||||
ndp *vip.NdpResponder
|
||||
mu sync.Mutex
|
||||
counter int
|
||||
}
|
||||
|
||||
@@ -31,7 +33,8 @@ func NewManager(config *kubevip.Config) *Manager {
|
||||
config.ArpBroadcastRate = 3000
|
||||
}
|
||||
return &Manager{
|
||||
config: config,
|
||||
instances: make(map[string]*Instance),
|
||||
config: config,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,19 +51,16 @@ func (i *Instance) Name() string {
|
||||
}
|
||||
|
||||
func (m *Manager) Insert(instance *Instance) {
|
||||
i, err := m.get(instance.Name())
|
||||
if err != nil {
|
||||
log.Error("[ARP manager] unable to insert instance", "err", err)
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
|
||||
existing := m.instances[instance.Name()]
|
||||
if existing == nil {
|
||||
m.instances[instance.Name()] = instance
|
||||
log.Info("[ARP manager] inserting ARP/NDP instance", "name", instance.Name())
|
||||
return
|
||||
}
|
||||
if i == nil {
|
||||
log.Info("[ARP manager] inserting ARP/NDP instance", "name", instance.Name())
|
||||
m.instances.Store(instance.Name(), instance)
|
||||
} else {
|
||||
i.mu.Lock()
|
||||
defer i.mu.Unlock()
|
||||
i.counter++
|
||||
}
|
||||
existing.counter++
|
||||
}
|
||||
|
||||
func (m *Manager) Remove(instance *Instance) {
|
||||
@@ -77,14 +77,11 @@ func (m *Manager) RemoveOnLeadershipLoss(instance *Instance) {
|
||||
}
|
||||
|
||||
func (m *Manager) RemoveWithIPDelete(instance *Instance, deleteIP bool) {
|
||||
i, err := m.get(instance.Name())
|
||||
if err != nil {
|
||||
log.Error("[ARP manager] unable to remove the instance", "err", err)
|
||||
return
|
||||
}
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
|
||||
i := m.instances[instance.Name()]
|
||||
if i != nil {
|
||||
i.mu.Lock()
|
||||
defer i.mu.Unlock()
|
||||
i.counter--
|
||||
if i.counter == 0 {
|
||||
log.Info("[ARP manager] removing ARP/NDP instance", "name", instance.Name())
|
||||
@@ -93,7 +90,7 @@ func (m *Manager) RemoveWithIPDelete(instance *Instance, deleteIP bool) {
|
||||
log.Error("failed to delete IP", "address", instance.network.IP(), "err", err)
|
||||
}
|
||||
}
|
||||
m.instances.Delete(instance.Name())
|
||||
delete(m.instances, instance.Name())
|
||||
}
|
||||
} else {
|
||||
log.Warn("[ARP manager] unable to remove the instance - instance not found", "name", instance.Name())
|
||||
@@ -101,14 +98,11 @@ func (m *Manager) RemoveWithIPDelete(instance *Instance, deleteIP bool) {
|
||||
}
|
||||
|
||||
func (m *Manager) Count(name string) int {
|
||||
i, err := m.get(name)
|
||||
if err != nil {
|
||||
log.Error("[ARP manager] unable to count instance", "err", err)
|
||||
return -1
|
||||
}
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
|
||||
i := m.instances[name]
|
||||
if i != nil {
|
||||
i.mu.Lock()
|
||||
defer i.mu.Unlock()
|
||||
return i.counter
|
||||
}
|
||||
return 0
|
||||
@@ -158,35 +152,22 @@ func (m *Manager) StartAdvertisement(ctx context.Context, killFunc func()) {
|
||||
case <-ctx.Done(): // if cancel() execute
|
||||
return
|
||||
case <-ticker.C: // send gratuitous ARP/NDP on each tick
|
||||
m.instances.Range(func(_ any, instance any) bool {
|
||||
if i, ok := instance.(*Instance); ok {
|
||||
i.mu.Lock()
|
||||
defer i.mu.Unlock()
|
||||
if i.counter > 0 {
|
||||
ensureIPAndSendGratuitous(i)
|
||||
} else {
|
||||
// this instance should not be advertised - delete the IP just in case...
|
||||
if _, err := i.network.DeleteIP(); err != nil {
|
||||
log.Error("[ARP manager] failed to delete IP", "address", i.network.IP(), "err", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
})
|
||||
m.advertiseAll()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (m *Manager) get(name string) (*Instance, error) {
|
||||
i, exists := m.instances.Load(name)
|
||||
if !exists {
|
||||
return nil, nil
|
||||
func (m *Manager) advertiseAll() {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
|
||||
for _, instance := range m.instances {
|
||||
if instance.counter > 0 {
|
||||
ensureIPAndSendGratuitous(instance)
|
||||
} else if _, err := instance.network.DeleteIP(); err != nil {
|
||||
log.Error("[ARP manager] failed to delete IP", "address", instance.network.IP(), "err", err)
|
||||
}
|
||||
}
|
||||
inst, ok := i.(*Instance)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("value for name %q is not of Instance pointer type", name)
|
||||
}
|
||||
return inst, nil
|
||||
}
|
||||
|
||||
// ensureIPAndSendGratuitous - adds IP to the interface if missing, and send
|
||||
@@ -255,13 +236,18 @@ func watch(ctx context.Context, interfaceName string, operStateHandler func(netl
|
||||
return fmt.Errorf("interface %s is not physical, ignoring", interfaceName)
|
||||
}
|
||||
|
||||
events := make(chan netlink.LinkUpdate)
|
||||
// The subscription is buffered and drained on exit: netlink parks its reader
|
||||
// goroutine on an unread send, which closing done alone does not release.
|
||||
events := make(chan netlink.LinkUpdate, linkSubscriptionBuffer)
|
||||
done := make(chan struct{})
|
||||
|
||||
if err := netlink.LinkSubscribe(events, done); err != nil {
|
||||
return fmt.Errorf("failed to subscribe to the interface events: %w", err)
|
||||
}
|
||||
defer close(done)
|
||||
defer func() {
|
||||
close(done)
|
||||
drainLinkUpdates(events)
|
||||
}()
|
||||
|
||||
// handle initial state
|
||||
operStateHandler(ifname.Attrs().OperState)
|
||||
@@ -290,3 +276,21 @@ func watch(ctx context.Context, interfaceName string, operStateHandler func(netl
|
||||
func isUp(operState netlink.LinkOperState) bool {
|
||||
return operState == netlink.OperUp
|
||||
}
|
||||
|
||||
// drainLinkUpdates releases a netlink sender that is parked on an unread update
|
||||
// so its goroutine can observe the closed subscription and exit.
|
||||
func drainLinkUpdates(events <-chan netlink.LinkUpdate) {
|
||||
timer := time.NewTimer(100 * time.Millisecond)
|
||||
defer timer.Stop()
|
||||
|
||||
for {
|
||||
select {
|
||||
case _, ok := <-events:
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
case <-timer.C:
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
119
pkg/arp/arp_test.go
Normal file
119
pkg/arp/arp_test.go
Normal file
@@ -0,0 +1,119 @@
|
||||
package arp
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/vishvananda/netlink"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
)
|
||||
|
||||
func TestDrainLinkUpdatesReleasesParkedSender(t *testing.T) {
|
||||
events := make(chan netlink.LinkUpdate)
|
||||
sent := make(chan struct{})
|
||||
go func() {
|
||||
events <- netlink.LinkUpdate{}
|
||||
close(sent)
|
||||
}()
|
||||
|
||||
drainLinkUpdates(events)
|
||||
|
||||
select {
|
||||
case <-sent:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("netlink sender is still parked on an unread link update")
|
||||
}
|
||||
}
|
||||
|
||||
// stubNetwork is a minimal vip.Network implementation; only ARPName matters here.
|
||||
type stubNetwork struct {
|
||||
name string
|
||||
deleteStarted chan struct{}
|
||||
releaseDelete chan struct{}
|
||||
}
|
||||
|
||||
func (s *stubNetwork) AddIP(bool, bool, ...int) (bool, error) { return false, nil }
|
||||
func (s *stubNetwork) AddRoute(bool) (bool, error) { return false, nil }
|
||||
func (s *stubNetwork) ReplaceRoute() error { return nil }
|
||||
func (s *stubNetwork) DeleteIP() (bool, error) {
|
||||
if s.deleteStarted != nil {
|
||||
close(s.deleteStarted)
|
||||
<-s.releaseDelete
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
func (s *stubNetwork) DeleteRoute() error { return nil }
|
||||
func (s *stubNetwork) UpdateRoutes() (bool, error) { return false, nil }
|
||||
func (s *stubNetwork) IsSet() (*netlink.Addr, error) { return nil, nil }
|
||||
func (s *stubNetwork) IP() string { return "" }
|
||||
func (s *stubNetwork) CIDR() string { return "" }
|
||||
func (s *stubNetwork) IPisLinkLocal() bool { return false }
|
||||
func (s *stubNetwork) PrepareRoute() *netlink.Route { return nil }
|
||||
func (s *stubNetwork) RouteHash() string { return "" }
|
||||
func (s *stubNetwork) SetIP(string) error { return nil }
|
||||
func (s *stubNetwork) SetServicePorts(*v1.Service) {}
|
||||
func (s *stubNetwork) Interface() string { return "eth0" }
|
||||
func (s *stubNetwork) IsDADFAIL() bool { return false }
|
||||
func (s *stubNetwork) IsDNS() bool { return false }
|
||||
func (s *stubNetwork) IsDDNS() bool { return false }
|
||||
func (s *stubNetwork) DDNSHostName() string { return "" }
|
||||
func (s *stubNetwork) DNSName() string { return "" }
|
||||
func (s *stubNetwork) SetMask(string) error { return nil }
|
||||
func (s *stubNetwork) SetHasEndpoints(bool) {}
|
||||
func (s *stubNetwork) HasEndpoints() bool { return false }
|
||||
func (s *stubNetwork) ARPName() string { return s.name }
|
||||
func (s *stubNetwork) GetPossibleSubnets() string { return "" }
|
||||
func (s *stubNetwork) DHCPFamily() string { return "" }
|
||||
func (s *stubNetwork) IPVSMark() uint32 { return 0 }
|
||||
|
||||
// TestManagerInsertConcurrentFirstRegistrationsDoNotLoseClaims guards the
|
||||
// get-then-store race: two never-before-seen instances for the same ARP name
|
||||
// registering concurrently must both be counted, not just the last writer.
|
||||
func TestManagerInsertConcurrentFirstRegistrationsDoNotLoseClaims(t *testing.T) {
|
||||
m := NewManager(&kubevip.Config{ArpBroadcastRate: 3000})
|
||||
const concurrent = 8
|
||||
|
||||
var wg sync.WaitGroup
|
||||
for range concurrent {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
m.Insert(NewInstance(&stubNetwork{name: "shared"}, nil))
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
if got := m.Count("shared"); got != concurrent {
|
||||
t.Fatalf("Count() = %d, want %d claims registered", got, concurrent)
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerInsertDoesNotJoinEntryBeingRemoved(t *testing.T) {
|
||||
m := NewManager(&kubevip.Config{ArpBroadcastRate: 3000})
|
||||
deleteStarted := make(chan struct{})
|
||||
releaseDelete := make(chan struct{})
|
||||
first := NewInstance(&stubNetwork{name: "shared", deleteStarted: deleteStarted, releaseDelete: releaseDelete}, nil)
|
||||
m.Insert(first)
|
||||
|
||||
removeDone := make(chan struct{})
|
||||
go func() {
|
||||
m.Remove(first)
|
||||
close(removeDone)
|
||||
}()
|
||||
<-deleteStarted
|
||||
|
||||
insertDone := make(chan struct{})
|
||||
go func() {
|
||||
m.Insert(NewInstance(&stubNetwork{name: "shared"}, nil))
|
||||
close(insertDone)
|
||||
}()
|
||||
close(releaseDelete)
|
||||
<-removeDone
|
||||
<-insertDone
|
||||
|
||||
if got := m.Count("shared"); got != 1 {
|
||||
t.Fatalf("Count() = %d, want replacement claim registered", got)
|
||||
}
|
||||
}
|
||||
@@ -47,10 +47,10 @@ func (e *Entry) Check() bool {
|
||||
// homeConfigPath := filepath.Join(os.Getenv("HOME"), ".kube", "config")
|
||||
|
||||
var k8sAddr string
|
||||
if utils.IsIPv4(e.Addr) {
|
||||
k8sAddr = fmt.Sprintf("%s:%v", e.Addr, e.Port)
|
||||
} else {
|
||||
if utils.IsIPv6(e.Addr) {
|
||||
k8sAddr = fmt.Sprintf("[%s]:%v", e.Addr, e.Port)
|
||||
} else {
|
||||
k8sAddr = fmt.Sprintf("%s:%v", e.Addr, e.Port)
|
||||
}
|
||||
|
||||
switch {
|
||||
|
||||
126
pkg/bgp/peers.go
126
pkg/bgp/peers.go
@@ -3,7 +3,7 @@ package bgp
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
log "log/slog"
|
||||
"net"
|
||||
"net/netip"
|
||||
"strconv"
|
||||
@@ -25,6 +25,11 @@ const defaultBGPPort uint32 = 179
|
||||
|
||||
// AddPeer will add peers to the BGP configuration
|
||||
func (b *Server) AddPeer(ctx context.Context, peer kubevip.BGPPeer) (err error) {
|
||||
remotePort := defaultBGPPort
|
||||
if peer.Port != 0 {
|
||||
remotePort = uint32(peer.Port)
|
||||
}
|
||||
|
||||
p := &api.Peer{
|
||||
Conf: &api.PeerConf{
|
||||
NeighborAddress: peer.Address,
|
||||
@@ -50,7 +55,7 @@ func (b *Server) AddPeer(ctx context.Context, peer kubevip.BGPPeer) (err error)
|
||||
Transport: &api.Transport{
|
||||
MtuDiscovery: true,
|
||||
RemoteAddress: peer.Address,
|
||||
RemotePort: defaultBGPPort,
|
||||
RemotePort: remotePort,
|
||||
},
|
||||
}
|
||||
|
||||
@@ -75,77 +80,88 @@ func (b *Server) AddPeer(ctx context.Context, peer kubevip.BGPPeer) (err error)
|
||||
}
|
||||
}
|
||||
|
||||
if b.c.MpbgpNexthop != "" {
|
||||
p.AfiSafis = []*api.AfiSafi{
|
||||
{
|
||||
Config: &api.AfiSafiConfig{
|
||||
Family: &api.Family{
|
||||
Afi: api.Family_AFI_IP,
|
||||
Safi: api.Family_SAFI_UNICAST,
|
||||
},
|
||||
Enabled: true,
|
||||
},
|
||||
},
|
||||
{
|
||||
Config: &api.AfiSafiConfig{
|
||||
Family: &api.Family{
|
||||
Afi: api.Family_AFI_IP6,
|
||||
Safi: api.Family_SAFI_UNICAST,
|
||||
},
|
||||
Enabled: true,
|
||||
},
|
||||
},
|
||||
}
|
||||
mpBGP := b.c.MpbgpNexthop
|
||||
|
||||
peer.SetMpbgpOptions(b.c)
|
||||
if peer.MpbgpNexthop != "" {
|
||||
mpBGP = peer.MpbgpNexthop
|
||||
}
|
||||
|
||||
if mpBGP != "" {
|
||||
ipv4Address, ipv6Address, err := peer.FindMpbgpAddresses(p, b.c)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to get MP-BGP addresses: %w", err)
|
||||
}
|
||||
log.Error("failed to get MP-BGP addresses, will not us MP-BGP for this host", "error", err)
|
||||
b.setPeerSource(p)
|
||||
} else {
|
||||
p.AfiSafis = []*api.AfiSafi{
|
||||
{
|
||||
Config: &api.AfiSafiConfig{
|
||||
Family: &api.Family{
|
||||
Afi: api.Family_AFI_IP,
|
||||
Safi: api.Family_SAFI_UNICAST,
|
||||
},
|
||||
Enabled: true,
|
||||
},
|
||||
},
|
||||
{
|
||||
Config: &api.AfiSafiConfig{
|
||||
Family: &api.Family{
|
||||
Afi: api.Family_AFI_IP6,
|
||||
Safi: api.Family_SAFI_UNICAST,
|
||||
},
|
||||
Enabled: true,
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
mask := strconv.Itoa(vip.DefaultMaskIPv6)
|
||||
address := ipv4Address
|
||||
family := api.Family_AFI_IP
|
||||
if utils.IsIPv4(p.Conf.NeighborAddress) {
|
||||
mask = strconv.Itoa(vip.DefaultMaskIPv4)
|
||||
address = ipv6Address
|
||||
family = api.Family_AFI_IP6
|
||||
}
|
||||
peer.SetMpbgpOptions(b.c)
|
||||
|
||||
err = b.s.AddDefinedSet(ctx, &api.AddDefinedSetRequest{
|
||||
DefinedSet: &api.DefinedSet{
|
||||
DefinedType: api.DefinedType_DEFINED_TYPE_NEIGHBOR,
|
||||
Name: fmt.Sprintf("peer-%s", p.Conf.NeighborAddress),
|
||||
List: []string{fmt.Sprintf("%s/%s", p.Conf.NeighborAddress, mask)},
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to add defined set: %v", err)
|
||||
}
|
||||
mask := strconv.Itoa(vip.DefaultMaskIPv6)
|
||||
address := ipv4Address
|
||||
family := api.Family_AFI_IP
|
||||
if utils.IsIPv4(p.Conf.NeighborAddress) {
|
||||
mask = strconv.Itoa(vip.DefaultMaskIPv4)
|
||||
address = ipv6Address
|
||||
family = api.Family_AFI_IP6
|
||||
}
|
||||
|
||||
if address != "" {
|
||||
if err := insertPolicy(ctx, b.s, address, p, family); err != nil {
|
||||
return fmt.Errorf("failed to add policy: %w", err)
|
||||
err = b.s.AddDefinedSet(ctx, &api.AddDefinedSetRequest{
|
||||
DefinedSet: &api.DefinedSet{
|
||||
DefinedType: api.DefinedType_DEFINED_TYPE_NEIGHBOR,
|
||||
Name: fmt.Sprintf("peer-%s", p.Conf.NeighborAddress),
|
||||
List: []string{fmt.Sprintf("%s/%s", p.Conf.NeighborAddress, mask)},
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to add defined set: %v", err)
|
||||
}
|
||||
|
||||
if address != "" {
|
||||
if err := insertPolicy(ctx, b.s, address, p, family); err != nil {
|
||||
return fmt.Errorf("failed to add policy: %w", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if b.c.SourceIP != "" {
|
||||
p.Transport.LocalAddress = b.c.SourceIP
|
||||
}
|
||||
|
||||
if b.c.SourceIF != "" {
|
||||
p.Transport.BindInterface = b.c.SourceIF
|
||||
}
|
||||
b.setPeerSource(p)
|
||||
}
|
||||
|
||||
if err := b.s.AddPeer(ctx, &api.AddPeerRequest{Peer: p}); err != nil {
|
||||
return fmt.Errorf("failed to add peer: %v", err)
|
||||
}
|
||||
slog.Info("[BGP]", "peer", p.Conf.NeighborAddress, "AS", p.Conf.PeerAsn, "BFD", p.Bfd)
|
||||
log.Info("[BGP]", "peer", p.Conf.NeighborAddress, "AS", p.Conf.PeerAsn, "BFD", p.Bfd)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (b *Server) setPeerSource(p *api.Peer) {
|
||||
if b.c.SourceIP != "" {
|
||||
p.Transport.LocalAddress = b.c.SourceIP
|
||||
}
|
||||
|
||||
if b.c.SourceIF != "" {
|
||||
p.Transport.BindInterface = b.c.SourceIF
|
||||
}
|
||||
}
|
||||
|
||||
func (b *Server) getPath(ip net.IP) *apiutil.Path {
|
||||
isV6 := ip.To4() == nil
|
||||
|
||||
|
||||
146
pkg/bgp/peers_config_test.go
Normal file
146
pkg/bgp/peers_config_test.go
Normal file
@@ -0,0 +1,146 @@
|
||||
package bgp
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
log "log/slog"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
api "github.com/osrg/gobgp/v4/api"
|
||||
gobgp "github.com/osrg/gobgp/v4/pkg/server"
|
||||
)
|
||||
|
||||
func TestAddPeerConfiguresTransportOptions(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
newServer func(*testing.T) *Server
|
||||
peer kubevip.BGPPeer
|
||||
wantPort uint32
|
||||
wantLocalAddr string
|
||||
wantInterface string
|
||||
}{
|
||||
{
|
||||
name: "configured remote port",
|
||||
newServer: func(t *testing.T) *Server {
|
||||
return newStartedTestBGPServer(t, kubevip.BGPConfig{
|
||||
AS: 65000,
|
||||
RouterID: "192.0.2.1",
|
||||
Peers: []kubevip.BGPPeer{{Address: "192.0.2.10", AS: 65001}},
|
||||
})
|
||||
},
|
||||
peer: kubevip.BGPPeer{Address: "192.0.2.10", AS: 65001, Port: 180},
|
||||
wantPort: 180,
|
||||
},
|
||||
{
|
||||
name: "configured source interface after MP-BGP fallback",
|
||||
newServer: func(t *testing.T) *Server {
|
||||
return newPeerTestServer(t, kubevip.BGPConfig{
|
||||
AS: 65000,
|
||||
RouterID: "192.0.2.1",
|
||||
SourceIF: "lo",
|
||||
MpbgpNexthop: "fixed",
|
||||
Peers: []kubevip.BGPPeer{{Address: "192.0.2.20", AS: 65001}},
|
||||
MpbgpIPv4: "",
|
||||
MpbgpIPv6: "",
|
||||
})
|
||||
},
|
||||
peer: kubevip.BGPPeer{Address: "192.0.2.20", AS: 65001},
|
||||
wantInterface: "lo",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
tt := tt
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
server := tt.newServer(t)
|
||||
if err := server.AddPeer(context.Background(), tt.peer); err != nil {
|
||||
t.Fatalf("AddPeer() error = %v", err)
|
||||
}
|
||||
|
||||
peer := listTestPeer(t, server, tt.peer.Address)
|
||||
if peer.GetTransport() == nil {
|
||||
t.Fatal("configured peer has no transport")
|
||||
}
|
||||
if tt.wantPort != 0 && peer.GetTransport().GetRemotePort() != tt.wantPort {
|
||||
t.Fatalf("remote port = %d, want %d", peer.GetTransport().GetRemotePort(), tt.wantPort)
|
||||
}
|
||||
if tt.wantLocalAddr != "" && peer.GetTransport().GetLocalAddress() != tt.wantLocalAddr {
|
||||
t.Fatalf("local address = %q, want %q", peer.GetTransport().GetLocalAddress(), tt.wantLocalAddr)
|
||||
}
|
||||
if tt.wantInterface != "" && peer.GetTransport().GetBindInterface() != tt.wantInterface {
|
||||
t.Fatalf("bind interface = %q, want %q", peer.GetTransport().GetBindInterface(), tt.wantInterface)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func newStartedTestBGPServer(t *testing.T, config kubevip.BGPConfig) *Server {
|
||||
t.Helper()
|
||||
|
||||
server, err := NewBGPServer(config, log.LevelError)
|
||||
if err != nil {
|
||||
t.Fatalf("NewBGPServer() error = %v", err)
|
||||
}
|
||||
|
||||
go server.s.Serve()
|
||||
if err := server.s.StartBgp(context.Background(), &api.StartBgpRequest{
|
||||
Global: &api.Global{
|
||||
Asn: config.AS,
|
||||
RouterId: config.RouterID,
|
||||
ListenPort: -1,
|
||||
},
|
||||
}); err != nil {
|
||||
server.s.Stop()
|
||||
t.Fatalf("StartBgp() error = %v", err)
|
||||
}
|
||||
t.Cleanup(server.s.Stop)
|
||||
|
||||
return server
|
||||
}
|
||||
|
||||
func listTestPeer(t *testing.T, server *Server, address string) *api.Peer {
|
||||
t.Helper()
|
||||
|
||||
var got *api.Peer
|
||||
if err := server.s.ListPeer(context.Background(), &api.ListPeerRequest{Address: address}, func(peer *api.Peer) {
|
||||
got = peer
|
||||
}); err != nil {
|
||||
t.Fatalf("ListPeer() error = %v", err)
|
||||
}
|
||||
if got == nil {
|
||||
t.Fatalf("ListPeer() returned no peer for %s", address)
|
||||
}
|
||||
return got
|
||||
}
|
||||
|
||||
func newPeerTestServer(t *testing.T, cfg kubevip.BGPConfig) *Server {
|
||||
t.Helper()
|
||||
raw := startEmbeddedRawBGP(t)
|
||||
return &Server{s: raw, c: &cfg, tracker: make(map[string]map[string]bool)}
|
||||
}
|
||||
|
||||
func startEmbeddedRawBGP(t *testing.T) *gobgp.BgpServer {
|
||||
t.Helper()
|
||||
raw := gobgp.NewBgpServer()
|
||||
go raw.Serve()
|
||||
if err := raw.StartBgp(context.Background(), &api.StartBgpRequest{
|
||||
Global: &api.Global{
|
||||
Asn: 65000,
|
||||
RouterId: "192.0.2.1",
|
||||
ListenPort: -1,
|
||||
},
|
||||
}); err != nil {
|
||||
t.Fatalf("starting embedded BGP server: %v", err)
|
||||
}
|
||||
var stopOnce sync.Once
|
||||
t.Cleanup(func() {
|
||||
stopOnce.Do(func() {
|
||||
if err := raw.StopBgp(context.Background(), &api.StopBgpRequest{}); err != nil {
|
||||
t.Logf("stopping embedded BGP server: %v", err)
|
||||
}
|
||||
})
|
||||
})
|
||||
return raw
|
||||
}
|
||||
@@ -15,6 +15,11 @@ import (
|
||||
gobgp "github.com/osrg/gobgp/v4/pkg/server"
|
||||
)
|
||||
|
||||
type BGPManager interface {
|
||||
AddHost(ctx context.Context, addr string, object string) error
|
||||
DelHost(ctx context.Context, addr string, object string) error
|
||||
}
|
||||
|
||||
// Server manages a server object
|
||||
type Server struct {
|
||||
s *gobgp.BgpServer
|
||||
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
"fmt"
|
||||
"net/http"
|
||||
"os"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
log "log/slog"
|
||||
@@ -15,13 +16,14 @@ import (
|
||||
"github.com/kube-vip/kube-vip/pkg/networkinterface"
|
||||
"github.com/kube-vip/kube-vip/pkg/node"
|
||||
"github.com/kube-vip/kube-vip/pkg/route"
|
||||
"github.com/kube-vip/kube-vip/pkg/utils"
|
||||
"github.com/kube-vip/kube-vip/pkg/vip"
|
||||
)
|
||||
|
||||
// Cluster - The Cluster object manages the state of the cluster for a particular node
|
||||
type Cluster struct {
|
||||
stop chan bool
|
||||
stop chan struct{}
|
||||
stopMu sync.Mutex
|
||||
service *servicesWorker
|
||||
Network []vip.Network
|
||||
arpMgr *arp.Manager
|
||||
routeMgr *route.Manager
|
||||
@@ -30,6 +32,13 @@ type Cluster struct {
|
||||
healthCheckHTTPClient *http.Client
|
||||
}
|
||||
|
||||
type servicesWorker struct {
|
||||
stop chan struct{}
|
||||
done chan struct{}
|
||||
stopping bool
|
||||
preserveVIPs map[string]struct{}
|
||||
}
|
||||
|
||||
// InitCluster - Will attempt to initialise all of the required settings for the cluster
|
||||
func InitCluster(c *kubevip.Config, disableVIP bool, intfMgr *networkinterface.Manager, arpMgr *arp.Manager,
|
||||
routeMgr *route.Manager, nodeLabelMgr node.Labeler) (*Cluster, error) {
|
||||
@@ -56,7 +65,7 @@ func InitCluster(c *kubevip.Config, disableVIP bool, intfMgr *networkinterface.M
|
||||
newCluster := &Cluster{
|
||||
Network: networks,
|
||||
arpMgr: arpMgr,
|
||||
stop: make(chan bool),
|
||||
stop: make(chan struct{}),
|
||||
routeMgr: routeMgr,
|
||||
nodeLabelMgr: nodeLabelMgr,
|
||||
healthCheckHTTPClient: healthCheckHTTPClient,
|
||||
@@ -81,7 +90,7 @@ func startNetworking(c *kubevip.Config, intfMgr *networkinterface.Manager) ([]vi
|
||||
network, err := vip.NewConfig(addr, c.Interface, c.LoInterfaceGlobalScope, c.VIPSubnet, c.DDNS, c.DHCPMode,
|
||||
c.RequireDualStack, c.IsDualStack, c.RoutingTableID, c.RoutingTableType, c.RoutingProtocol, c.DNSMode,
|
||||
c.LoadBalancerForwardingMethod, c.IptablesBackend, c.EnableLoadBalancer, c.LoadBalancerPort,
|
||||
c.EnableServiceSecurity, intfMgr, c.EgressWithNftables)
|
||||
c.EnableServiceSecurity, intfMgr, c.EgressWithNftables, c.SkipDAD)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -93,11 +102,115 @@ func startNetworking(c *kubevip.Config, intfMgr *networkinterface.Manager) ([]vi
|
||||
|
||||
// Stop - Will stop the Cluster and release VIP if needed
|
||||
func (cluster *Cluster) Stop() {
|
||||
// Close the stop channel, which will shut down the VIP (if needed)
|
||||
if cluster.stop != nil {
|
||||
close(cluster.stop)
|
||||
cluster.stop = make(chan bool) // recreate channel for future use
|
||||
cluster.stopMu.Lock()
|
||||
defer cluster.stopMu.Unlock()
|
||||
if cluster.service != nil {
|
||||
workers := cluster.service
|
||||
if workers.stopping {
|
||||
return
|
||||
}
|
||||
workers.stopping = true
|
||||
cluster.stop = make(chan struct{})
|
||||
close(workers.stop)
|
||||
return
|
||||
}
|
||||
stop := cluster.stop
|
||||
cluster.stop = make(chan struct{})
|
||||
close(stop)
|
||||
}
|
||||
|
||||
// StopAndWait signals the current Service worker generation and waits until it
|
||||
// has finished its datapath cleanup.
|
||||
func (cluster *Cluster) StopAndWait() {
|
||||
cluster.stopAndWait(nil)
|
||||
}
|
||||
|
||||
// StopAndWaitPreserving stops the current Service worker generation while
|
||||
// preserving the supplied VIPs for another Service that shares the same lease.
|
||||
func (cluster *Cluster) StopAndWaitPreserving(addresses ...string) {
|
||||
preserve := make(map[string]struct{}, len(addresses))
|
||||
for _, address := range addresses {
|
||||
preserve[address] = struct{}{}
|
||||
}
|
||||
cluster.stopAndWait(preserve)
|
||||
}
|
||||
|
||||
func (cluster *Cluster) stopAndWait(preserveVIPs map[string]struct{}) {
|
||||
workers, signal := cluster.prepareServiceStop(preserveVIPs)
|
||||
if workers == nil {
|
||||
return
|
||||
}
|
||||
if signal {
|
||||
close(workers.stop)
|
||||
}
|
||||
<-workers.done
|
||||
}
|
||||
|
||||
func (cluster *Cluster) prepareServiceStop(preserveVIPs map[string]struct{}) (*servicesWorker, bool) {
|
||||
cluster.stopMu.Lock()
|
||||
defer cluster.stopMu.Unlock()
|
||||
workers := cluster.service
|
||||
if workers == nil {
|
||||
return nil, false
|
||||
}
|
||||
if workers.stopping {
|
||||
workers.preserveVIPs = mergeVIPs(workers.preserveVIPs, preserveVIPs)
|
||||
return workers, false
|
||||
}
|
||||
workers.stopping = true
|
||||
workers.preserveVIPs = preserveVIPs
|
||||
cluster.stop = make(chan struct{})
|
||||
return workers, true
|
||||
}
|
||||
|
||||
func (cluster *Cluster) startServicesWorker() (<-chan struct{}, chan struct{}, error) {
|
||||
cluster.stopMu.Lock()
|
||||
defer cluster.stopMu.Unlock()
|
||||
if cluster.service != nil {
|
||||
return nil, nil, fmt.Errorf("load balancer workers already running")
|
||||
}
|
||||
workers := &servicesWorker{stop: cluster.stop, done: make(chan struct{})}
|
||||
cluster.service = workers
|
||||
return workers.stop, workers.done, nil
|
||||
}
|
||||
|
||||
func (cluster *Cluster) preserveServiceVIP(done chan struct{}, address string) bool {
|
||||
cluster.stopMu.Lock()
|
||||
defer cluster.stopMu.Unlock()
|
||||
if cluster.service == nil || cluster.service.done != done {
|
||||
return false
|
||||
}
|
||||
_, preserve := cluster.service.preserveVIPs[address]
|
||||
return preserve
|
||||
}
|
||||
|
||||
func mergeVIPs(existing, addresses map[string]struct{}) map[string]struct{} {
|
||||
if len(addresses) == 0 {
|
||||
return existing
|
||||
}
|
||||
if existing == nil {
|
||||
existing = make(map[string]struct{}, len(addresses))
|
||||
}
|
||||
for address := range addresses {
|
||||
existing[address] = struct{}{}
|
||||
}
|
||||
return existing
|
||||
}
|
||||
|
||||
func (cluster *Cluster) finishServicesWorker(done chan struct{}) {
|
||||
cluster.stopMu.Lock()
|
||||
defer cluster.stopMu.Unlock()
|
||||
if cluster.service == nil || cluster.service.done != done {
|
||||
return
|
||||
}
|
||||
cluster.service = nil
|
||||
close(done)
|
||||
}
|
||||
|
||||
func (cluster *Cluster) StopChannel() <-chan struct{} {
|
||||
cluster.stopMu.Lock()
|
||||
defer cluster.stopMu.Unlock()
|
||||
return cluster.stop
|
||||
}
|
||||
|
||||
func newHealthCheckHTTPClient(c *kubevip.Config) (*http.Client, error) {
|
||||
@@ -135,38 +248,41 @@ func newHealthCheckHTTPClient(c *kubevip.Config) (*http.Client, error) {
|
||||
}, nil
|
||||
}
|
||||
|
||||
// cleanupVIPs handles VIP removal based on the PreserveVIPOnLeadershipLoss configuration.
|
||||
// When preservation is enabled, IPv6 VIPs are always removed immediately to prevent DAD
|
||||
// failures on the new leader, while IPv4 VIPs are intentionally left in place.
|
||||
// When preservation is disabled (legacy behavior), all VIPs are removed.
|
||||
// cleanupVIPs releases the control plane VIPs after leadership was lost.
|
||||
// Nothing waits for the control plane layer2Update goroutine to observe the
|
||||
// cancelled context, so this caller usually still holds its own ARP claim and
|
||||
// has to delete the address itself.
|
||||
func (cluster *Cluster) cleanupVIPs(c *kubevip.Config) {
|
||||
for i := range cluster.Network {
|
||||
if c.EnableARP && cluster.arpMgr.Count(cluster.Network[i].ARPName()) > 1 {
|
||||
continue
|
||||
}
|
||||
|
||||
if c.PreserveVIPOnLeadershipLoss {
|
||||
if utils.IsIPv6(cluster.Network[i].IP()) {
|
||||
log.Info("[VIP] Removing IPv6 VIP immediately (required to prevent DAD failures on new leader)", "ip", cluster.Network[i].IP())
|
||||
deleted, err := cluster.Network[i].DeleteIP()
|
||||
if err != nil {
|
||||
log.Warn(err.Error())
|
||||
}
|
||||
if deleted {
|
||||
log.Info("deleted address", "IP", cluster.Network[i].IP(), "interface", cluster.Network[i].Interface())
|
||||
}
|
||||
} else {
|
||||
log.Info("[VIP] Preserving IPv4 VIP address on interface, only stopped ARP broadcasting", "ip", cluster.Network[i].IP())
|
||||
}
|
||||
} else {
|
||||
log.Info("[VIP] Deleting VIP", "ip", cluster.Network[i].IP())
|
||||
deleted, err := cluster.Network[i].DeleteIP()
|
||||
if err != nil {
|
||||
log.Warn(err.Error())
|
||||
}
|
||||
if deleted {
|
||||
log.Info("deleted address", "IP", cluster.Network[i].IP(), "interface", cluster.Network[i].Interface())
|
||||
}
|
||||
}
|
||||
cluster.cleanupVIP(c, cluster.Network[i], 1)
|
||||
}
|
||||
}
|
||||
|
||||
// cleanupServiceVIPs releases the service VIPs once the services worker has
|
||||
// drained. layer2Update already removed this instance's own claim by then, so
|
||||
// any remaining claim belongs to another service sharing the VIP.
|
||||
func (cluster *Cluster) cleanupServiceVIPs(c *kubevip.Config, done chan struct{}) {
|
||||
for i := range cluster.Network {
|
||||
if cluster.preserveServiceVIP(done, cluster.Network[i].IP()) {
|
||||
continue
|
||||
}
|
||||
cluster.cleanupVIP(c, cluster.Network[i], 0)
|
||||
}
|
||||
}
|
||||
|
||||
// cleanupVIP deletes the VIP unless somebody else still advertises it.
|
||||
// ownClaims is the number of ARP claims the caller may still hold itself.
|
||||
func (cluster *Cluster) cleanupVIP(c *kubevip.Config, network vip.Network, ownClaims int) {
|
||||
if c.EnableARP && cluster.arpMgr.Count(network.ARPName()) > ownClaims {
|
||||
return
|
||||
}
|
||||
|
||||
log.Info("[VIP] Deleting VIP", "ip", network.IP())
|
||||
deleted, err := network.DeleteIP()
|
||||
if err != nil {
|
||||
log.Warn(err.Error())
|
||||
}
|
||||
if deleted {
|
||||
log.Info("deleted address", "IP", network.IP(), "interface", network.Interface())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,6 +10,7 @@ import (
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
"github.com/kube-vip/kube-vip/pkg/utils"
|
||||
"github.com/kube-vip/kube-vip/pkg/vip"
|
||||
|
||||
log "log/slog"
|
||||
)
|
||||
@@ -25,36 +26,22 @@ func (cluster *Cluster) StartCluster(ctx context.Context, c *kubevip.Config,
|
||||
log.Info("cluster membership", "namespace", leaseID.Namespace(), "lock", leaseID.Name(), "id", c.NodeName)
|
||||
|
||||
objectName := lease.ObjectName(leaseID, "cp")
|
||||
objLease := leaseMgr.Add(ctx, leaseID)
|
||||
isNew := objLease.Add(objectName)
|
||||
objLease, _ := leaseMgr.Acquire(context.Background(), leaseID, objectName)
|
||||
defer leaseMgr.Delete(leaseID, objectName, objLease)
|
||||
|
||||
wg := sync.WaitGroup{}
|
||||
defer wg.Wait()
|
||||
|
||||
// Start a goroutine that will delete the lease when the service context is cancelled.
|
||||
// This is important for proper cleanup when a service is deleted - it ensures that
|
||||
// the lease context (svcLease.Ctx) gets cancelled, which causes RunOrDie to return.
|
||||
// Without this, RunOrDie would continue running until leadership is naturally lost.
|
||||
wg.Go(func() {
|
||||
<-objLease.Ctx.Done()
|
||||
leaseMgr.Delete(leaseID, objectName)
|
||||
})
|
||||
|
||||
if !isNew {
|
||||
log.Debug("this election was already done, waiting for it to finish", "lease", leaseName)
|
||||
<-objLease.Ctx.Done()
|
||||
return nil
|
||||
}
|
||||
electionCtx, cancelElection := objLease.NewElectionContext(ctx)
|
||||
defer cancelElection()
|
||||
|
||||
stop := cluster.StopChannel()
|
||||
wg.Go(func() {
|
||||
select {
|
||||
case <-cluster.stop:
|
||||
case <-ctx.Done():
|
||||
case <-stop:
|
||||
cancelElection()
|
||||
case <-electionCtx.Done():
|
||||
}
|
||||
|
||||
log.Info("Received termination, signaling cluster shutdown")
|
||||
// Cancel the leader context, which will in turn cancel the leadership
|
||||
objLease.Cancel()
|
||||
})
|
||||
|
||||
// (attempt to) Remove the virtual IP, in case it already exists
|
||||
@@ -69,54 +56,59 @@ func (cluster *Cluster) StartCluster(ctx context.Context, c *kubevip.Config,
|
||||
}
|
||||
}
|
||||
|
||||
objLease.Lock()
|
||||
for {
|
||||
if !objLease.BeginElection() {
|
||||
log.Debug("this election was already done, shared lease", "lease", leaseName)
|
||||
leaderGeneration, elected := objLease.WaitForLeaderGeneration(electionCtx)
|
||||
if !elected {
|
||||
if electionCtx.Err() != nil {
|
||||
return nil
|
||||
}
|
||||
// The runner that owned this shared lease's election ended it
|
||||
// without ever being elected; take over the campaign ourselves
|
||||
// instead of leaving the lease without an active runner.
|
||||
continue
|
||||
}
|
||||
|
||||
defer func() {
|
||||
objLease.Unlock()
|
||||
}()
|
||||
leaderCtx, cancelLeader := context.WithCancel(electionCtx)
|
||||
leaderWG := sync.WaitGroup{}
|
||||
leaderWG.Go(func() {
|
||||
cluster.OnStartedLeading(leaderCtx, c, em, bgpServer, killFunc, true)
|
||||
})
|
||||
|
||||
// this object is sharing lease with another object
|
||||
if objLease.Elected.Load() {
|
||||
log.Debug("this election was already done, shared lease", "lease", leaseName)
|
||||
// wait for leader election to start or context to be done
|
||||
select {
|
||||
case <-objLease.Started:
|
||||
case <-objLease.Ctx.Done():
|
||||
// Lease was cancelled (e.g., leader election ended), return immediately
|
||||
// This allows the restart loop to create a fresh lease
|
||||
log.Debug("lease context cancelled before leader election started", "lease", leaseName)
|
||||
return fmt.Errorf("lease %q context cancelled before leader election started", leaseName)
|
||||
log.Debug("cluster waiting for shared election to finish", "lease", leaseName)
|
||||
objLease.WaitForElectionEndAfter(electionCtx, leaderGeneration)
|
||||
cancelLeader()
|
||||
leaderWG.Wait()
|
||||
|
||||
cluster.OnStoppedLeading(c, bgpServer)
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
cluster.OnStartedLeading(c, objLease, em, bgpServer, killFunc, true)
|
||||
|
||||
log.Debug("cluster waiting for leader context done", "lease", leaseName)
|
||||
// wait for leaderelection to be finished
|
||||
<-objLease.Ctx.Done()
|
||||
|
||||
cluster.OnStoppedLeading(c, objLease, bgpServer)
|
||||
|
||||
return nil
|
||||
break
|
||||
}
|
||||
defer objLease.ElectionStopped()
|
||||
|
||||
run := &election.RunConfig{
|
||||
Config: c,
|
||||
LeaseID: leaseID,
|
||||
LeaseAnnotations: c.LeaseAnnotations,
|
||||
VIPs: controlPlaneElectionVIPs(c),
|
||||
Mgr: em,
|
||||
OnStartedLeading: func(context.Context) { //nolint TODO: potential clean code
|
||||
cluster.OnStartedLeading(c, objLease, em, bgpServer, killFunc, false)
|
||||
OnStartedLeading: func(ctx context.Context) {
|
||||
objLease.ElectionStarted()
|
||||
cluster.OnStartedLeading(ctx, c, em, bgpServer, killFunc, false)
|
||||
},
|
||||
OnStoppedLeading: func() {
|
||||
objLease.Elected.Store(false)
|
||||
cluster.OnStoppedLeading(c, objLease, bgpServer)
|
||||
objLease.ElectionStopped()
|
||||
cluster.OnStoppedLeading(c, bgpServer)
|
||||
},
|
||||
OnNewLeader: func(identity string) {
|
||||
cluster.OnNewLeader(identity, c)
|
||||
},
|
||||
}
|
||||
|
||||
if err := election.RunOrDie(objLease.Ctx, run, c); err != nil {
|
||||
if err := election.RunOrDie(electionCtx, run, c); err != nil {
|
||||
cluster.Stop()
|
||||
return fmt.Errorf("leaderelection failed: %w", err)
|
||||
}
|
||||
@@ -124,47 +116,31 @@ func (cluster *Cluster) StartCluster(ctx context.Context, c *kubevip.Config,
|
||||
return nil
|
||||
}
|
||||
|
||||
func (cluster *Cluster) OnStartedLeading(c *kubevip.Config, objLease *lease.Lease,
|
||||
em *election.Manager, bgpServer *bgp.Server, killFunc func(), isShared bool) {
|
||||
objLease.Elected.Store(true)
|
||||
objLease.Unlock()
|
||||
|
||||
// When we become leader, ensure we can take over VIPs even if they're preserved on other nodes
|
||||
if !isShared {
|
||||
close(objLease.Started)
|
||||
func controlPlaneElectionVIPs(config *kubevip.Config) []string {
|
||||
configured := config.VIP
|
||||
if config.Address != "" {
|
||||
configured = config.Address
|
||||
}
|
||||
return vip.Split(configured)
|
||||
}
|
||||
|
||||
func (cluster *Cluster) OnStartedLeading(ctx context.Context, c *kubevip.Config,
|
||||
em *election.Manager, bgpServer *bgp.Server, killFunc func(), _ bool) {
|
||||
labels := generateLabelsFromConfig(c.Address, kubevip.HasIP)
|
||||
if err := cluster.nodeLabelMgr.AddLabel(labels); err != nil {
|
||||
log.Error("error adding label to node", "err", err)
|
||||
}
|
||||
cluster.labelAdded = true
|
||||
|
||||
if c.PreserveVIPOnLeadershipLoss {
|
||||
log.Info("Becoming leader with VIP preservation enabled - ensuring VIP takeover")
|
||||
// Force add the VIPs (this will work even if they exist due to the precheck logic)
|
||||
for i := range cluster.Network {
|
||||
added, err := cluster.Network[i].AddIP(true, false)
|
||||
if err != nil {
|
||||
log.Error("failed to ensure VIP on leader takeover", "vip", cluster.Network[i].IP(), "err", err)
|
||||
} else if added {
|
||||
log.Info("took over VIP as new leader", "IP", cluster.Network[i].IP(), "interface", cluster.Network[i].Interface())
|
||||
} else {
|
||||
log.Info("VIP already configured on interface", "IP", cluster.Network[i].IP(), "interface", cluster.Network[i].Interface())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// As we're leading lets start the vip service
|
||||
err := cluster.StartVipService(objLease.Ctx, c, em, bgpServer, killFunc)
|
||||
err := cluster.StartVipService(ctx, c, em, bgpServer, killFunc)
|
||||
if err != nil {
|
||||
log.Error("starting VIP service on leader", "err", err)
|
||||
killFunc()
|
||||
}
|
||||
}
|
||||
|
||||
func (cluster *Cluster) OnStoppedLeading(c *kubevip.Config, objLease *lease.Lease,
|
||||
bgpServer *bgp.Server) {
|
||||
func (cluster *Cluster) OnStoppedLeading(c *kubevip.Config, bgpServer *bgp.Server) {
|
||||
// we can do cleanup here
|
||||
log.Info("This node is becoming a follower within the cluster")
|
||||
|
||||
@@ -176,9 +152,6 @@ func (cluster *Cluster) OnStoppedLeading(c *kubevip.Config, objLease *lease.Leas
|
||||
cluster.labelAdded = false
|
||||
}
|
||||
|
||||
// Stop the cluster context if it is running
|
||||
objLease.Cancel()
|
||||
|
||||
cluster.cleanupVIPs(c)
|
||||
|
||||
log.Error("lost leadership, restarting kube-vip")
|
||||
@@ -187,25 +160,6 @@ func (cluster *Cluster) OnStoppedLeading(c *kubevip.Config, objLease *lease.Leas
|
||||
func (cluster *Cluster) OnNewLeader(identity string, c *kubevip.Config) {
|
||||
// we're notified when new leader elected
|
||||
log.Info("New leader", "leader", identity)
|
||||
|
||||
// If we're not the new leader and we have VIPs preserved from previous leadership,
|
||||
// we need to clean them up to avoid conflicts.
|
||||
if identity != c.NodeName && c.PreserveVIPOnLeadershipLoss {
|
||||
log.Info("Cleaning up preserved VIPs as another node became leader", "new_leader", identity)
|
||||
for i := range cluster.Network {
|
||||
deleted, err := cluster.Network[i].DeleteIP()
|
||||
if err != nil {
|
||||
log.Warn("failed to cleanup preserved VIP", "vip", cluster.Network[i].IP(), "err", err)
|
||||
}
|
||||
if deleted {
|
||||
log.Info("cleaned up preserved VIP to avoid conflict", "IP", cluster.Network[i].IP(),
|
||||
"interface", cluster.Network[i].Interface(), "new_leader", identity)
|
||||
} else {
|
||||
log.Debug("VIP was not present on this node", "IP", cluster.Network[i].IP(),
|
||||
"interface", cluster.Network[i].Interface())
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func generateLabelsFromConfig(addr, labelKey string) map[string]string {
|
||||
|
||||
196
pkg/cluster/cluster_internal_test.go
Normal file
196
pkg/cluster/cluster_internal_test.go
Normal file
@@ -0,0 +1,196 @@
|
||||
package cluster
|
||||
|
||||
import (
|
||||
"context"
|
||||
"slices"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/arp"
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
"github.com/kube-vip/kube-vip/pkg/vip"
|
||||
"github.com/vishvananda/netlink"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
)
|
||||
|
||||
func TestControlPlaneElectionVIPsPreservesConfigOrder(t *testing.T) {
|
||||
config := &kubevip.Config{Address: "2001:db8::10,192.0.2.10"}
|
||||
want := []string{"2001:db8::10", "192.0.2.10"}
|
||||
if got := controlPlaneElectionVIPs(config); !slices.Equal(got, want) {
|
||||
t.Fatalf("controlPlaneElectionVIPs() = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
type recordingLabeler struct {
|
||||
added chan struct{}
|
||||
removed chan struct{}
|
||||
}
|
||||
|
||||
func (l *recordingLabeler) AddLabel(map[string]string) error {
|
||||
l.added <- struct{}{}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (l *recordingLabeler) RemoveLabel(map[string]string) error {
|
||||
l.removed <- struct{}{}
|
||||
return nil
|
||||
}
|
||||
|
||||
// stubNetwork is a minimal vip.Network implementation for exercising
|
||||
// cleanupVIP without a real interface.
|
||||
type stubNetwork struct {
|
||||
ip string
|
||||
deleteIPCalls int
|
||||
}
|
||||
|
||||
func (s *stubNetwork) AddIP(bool, bool, ...int) (bool, error) { return false, nil }
|
||||
func (s *stubNetwork) AddRoute(bool) (bool, error) { return false, nil }
|
||||
func (s *stubNetwork) ReplaceRoute() error { return nil }
|
||||
func (s *stubNetwork) DeleteIP() (bool, error) { s.deleteIPCalls++; return true, nil }
|
||||
func (s *stubNetwork) DeleteRoute() error { return nil }
|
||||
func (s *stubNetwork) UpdateRoutes() (bool, error) { return false, nil }
|
||||
func (s *stubNetwork) IsSet() (*netlink.Addr, error) { return nil, nil }
|
||||
func (s *stubNetwork) IP() string { return s.ip }
|
||||
func (s *stubNetwork) CIDR() string { return s.ip + "/32" }
|
||||
func (s *stubNetwork) IPisLinkLocal() bool { return false }
|
||||
func (s *stubNetwork) PrepareRoute() *netlink.Route { return nil }
|
||||
func (s *stubNetwork) RouteHash() string { return "" }
|
||||
func (s *stubNetwork) SetIP(string) error { return nil }
|
||||
func (s *stubNetwork) SetServicePorts(*v1.Service) {}
|
||||
func (s *stubNetwork) Interface() string { return "eth0" }
|
||||
func (s *stubNetwork) IsDADFAIL() bool { return false }
|
||||
func (s *stubNetwork) IsDNS() bool { return false }
|
||||
func (s *stubNetwork) IsDDNS() bool { return false }
|
||||
func (s *stubNetwork) DDNSHostName() string { return "" }
|
||||
func (s *stubNetwork) DNSName() string { return "" }
|
||||
func (s *stubNetwork) SetMask(string) error { return nil }
|
||||
func (s *stubNetwork) SetHasEndpoints(bool) {}
|
||||
func (s *stubNetwork) HasEndpoints() bool { return false }
|
||||
func (s *stubNetwork) ARPName() string { return "shared-vip" }
|
||||
func (s *stubNetwork) GetPossibleSubnets() string { return "" }
|
||||
func (s *stubNetwork) DHCPFamily() string { return "" }
|
||||
func (s *stubNetwork) IPVSMark() uint32 { return 0 }
|
||||
|
||||
// TestCleanupVIPRetainsSharedVIPWithOneSiblingLeft reproduces the off-by-one:
|
||||
// layer2Update already removes its own ARP claim before cleanupVIP runs, so a
|
||||
// single remaining sibling must still block deletion.
|
||||
func TestCleanupVIPRetainsSharedVIPWithOneSiblingLeft(t *testing.T) {
|
||||
arpMgr := arp.NewManager(&kubevip.Config{ArpBroadcastRate: 3000})
|
||||
netA := &stubNetwork{ip: "192.0.2.10"}
|
||||
netB := &stubNetwork{ip: "192.0.2.10"}
|
||||
|
||||
instA := arp.NewInstance(netA, nil)
|
||||
instB := arp.NewInstance(netB, nil)
|
||||
arpMgr.Insert(instA)
|
||||
arpMgr.Insert(instB)
|
||||
|
||||
// Cluster A's layer2Update goroutine ends first and drops its own claim,
|
||||
// leaving only sibling B registered.
|
||||
arpMgr.Remove(instA)
|
||||
|
||||
c := &Cluster{arpMgr: arpMgr}
|
||||
c.cleanupVIP(&kubevip.Config{EnableARP: true}, netA, 0)
|
||||
|
||||
if netA.deleteIPCalls != 0 {
|
||||
t.Fatalf("cleanupVIP deleted the shared VIP while a sibling was still registered")
|
||||
}
|
||||
}
|
||||
|
||||
// TestCleanupVIPsDeletesControlPlaneVIPHoldingItsOwnARPClaim covers the
|
||||
// leadership loss path: OnStoppedLeading runs concurrently with the control
|
||||
// plane layer2Update goroutine, so the VIP's only ARP claim is still the
|
||||
// caller's own and the address must still be removed before the process exits.
|
||||
func TestCleanupVIPsDeletesControlPlaneVIPHoldingItsOwnARPClaim(t *testing.T) {
|
||||
arpMgr := arp.NewManager(&kubevip.Config{ArpBroadcastRate: 3000})
|
||||
network := &stubNetwork{ip: "2001:db8::10"}
|
||||
arpMgr.Insert(arp.NewInstance(network, nil))
|
||||
|
||||
c := &Cluster{arpMgr: arpMgr, Network: []vip.Network{network}}
|
||||
c.cleanupVIPs(&kubevip.Config{EnableARP: true})
|
||||
|
||||
if network.deleteIPCalls != 1 {
|
||||
t.Fatalf("cleanupVIPs made %d DeleteIP calls, want 1", network.deleteIPCalls)
|
||||
}
|
||||
}
|
||||
|
||||
func TestControlPlaneFollowsSharedServiceElection(t *testing.T) {
|
||||
config := &kubevip.Config{KubernetesLeaderElection: kubevip.KubernetesLeaderElection{LeaseName: "default/shared"}}
|
||||
leaseID := lease.NewID(config.LeaderElectionType, "default", "shared")
|
||||
leaseMgr := lease.NewManager()
|
||||
sharedLease, _ := leaseMgr.Acquire(context.Background(), leaseID, "service")
|
||||
if !sharedLease.BeginElection() {
|
||||
t.Fatal("Service election did not start")
|
||||
}
|
||||
sharedLease.ElectionStarted()
|
||||
|
||||
labels := &recordingLabeler{added: make(chan struct{}, 1), removed: make(chan struct{}, 1)}
|
||||
cluster := &Cluster{stop: make(chan struct{}), nodeLabelMgr: labels}
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
done <- cluster.StartCluster(context.Background(), config, nil, nil, leaseMgr, func() {})
|
||||
}()
|
||||
select {
|
||||
case <-labels.added:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("control plane did not activate under the shared Service election")
|
||||
}
|
||||
|
||||
sharedLease.ElectionStopped()
|
||||
select {
|
||||
case err := <-done:
|
||||
if err != nil {
|
||||
t.Fatalf("shared control-plane follower returned an error: %v", err)
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("control plane did not stop after shared Service leadership ended")
|
||||
}
|
||||
select {
|
||||
case <-labels.removed:
|
||||
default:
|
||||
t.Fatal("control-plane label was not removed after shared leadership ended")
|
||||
}
|
||||
if sharedLease.Ctx.Err() != nil || leaseMgr.Get(leaseID) != sharedLease {
|
||||
t.Fatal("control-plane cleanup cancelled the surviving Service lease")
|
||||
}
|
||||
leaseMgr.Delete(leaseID, "service", sharedLease)
|
||||
}
|
||||
|
||||
func TestStopAndWaitPreservingUpgradesInProgressStop(t *testing.T) {
|
||||
done := make(chan struct{})
|
||||
service := &Cluster{
|
||||
stop: make(chan struct{}),
|
||||
service: &servicesWorker{
|
||||
stop: make(chan struct{}),
|
||||
done: done,
|
||||
stopping: true,
|
||||
},
|
||||
}
|
||||
|
||||
returned := make(chan struct{})
|
||||
go func() {
|
||||
service.StopAndWaitPreserving("192.0.2.10")
|
||||
close(returned)
|
||||
}()
|
||||
|
||||
deadline := time.Now().Add(time.Second)
|
||||
for {
|
||||
service.stopMu.Lock()
|
||||
_, preserving := service.service.preserveVIPs["192.0.2.10"]
|
||||
service.stopMu.Unlock()
|
||||
if preserving {
|
||||
break
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatal("preserving stop did not update the in-progress worker shutdown")
|
||||
}
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
|
||||
service.finishServicesWorker(done)
|
||||
select {
|
||||
case <-returned:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("preserving stop did not return after worker cleanup completed")
|
||||
}
|
||||
}
|
||||
32
pkg/cluster/cluster_stop_test.go
Normal file
32
pkg/cluster/cluster_stop_test.go
Normal file
@@ -0,0 +1,32 @@
|
||||
package cluster
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestStopConcurrentDoesNotRaceOrPanic(t *testing.T) {
|
||||
c := &Cluster{stop: make(chan struct{})}
|
||||
start := make(chan struct{})
|
||||
var wg sync.WaitGroup
|
||||
var panics atomic.Int64
|
||||
|
||||
for range 128 {
|
||||
wg.Go(func() {
|
||||
<-start
|
||||
defer func() {
|
||||
if recover() != nil {
|
||||
panics.Add(1)
|
||||
}
|
||||
}()
|
||||
c.Stop()
|
||||
})
|
||||
}
|
||||
|
||||
close(start)
|
||||
wg.Wait()
|
||||
if got := panics.Load(); got != 0 {
|
||||
t.Fatalf("concurrent Stop panicked %d time(s)", got)
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,8 @@ import (
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"syscall"
|
||||
@@ -27,14 +29,7 @@ import (
|
||||
"k8s.io/client-go/kubernetes"
|
||||
)
|
||||
|
||||
// BGPRouteManager allows to manage the routes announced by the BGP server.
|
||||
type BGPRouteManager interface {
|
||||
AddHost(ctx context.Context, addr string, object string) error
|
||||
DelHost(ctx context.Context, addr string, object string) error
|
||||
}
|
||||
|
||||
func (cluster *Cluster) StartVipService(ctx context.Context, c *kubevip.Config, em *election.Manager,
|
||||
bgpServer BGPRouteManager, killFunc func()) error {
|
||||
func (cluster *Cluster) StartVipService(ctx context.Context, c *kubevip.Config, em *election.Manager, bgpServer bgp.BGPManager, killFunc func()) error {
|
||||
|
||||
var err error
|
||||
|
||||
@@ -49,6 +44,9 @@ func (cluster *Cluster) StartVipService(ctx context.Context, c *kubevip.Config,
|
||||
loadbalancers := []*loadbalancer.IPVSLoadBalancer{}
|
||||
|
||||
for i := range cluster.Network {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
network := cluster.Network[i]
|
||||
|
||||
if network.IsDDNS() {
|
||||
@@ -111,7 +109,7 @@ func (cluster *Cluster) StartVipService(ctx context.Context, c *kubevip.Config,
|
||||
err = em.NodeWatcher(ctx, lb, c.Port)
|
||||
if err != nil {
|
||||
log.Error("Error watching node labels", "err", err)
|
||||
if errors.Is(err, &utils.PanicError{}) {
|
||||
if utils.IsPanicError(err) {
|
||||
killFunc()
|
||||
return
|
||||
}
|
||||
@@ -146,29 +144,41 @@ func (cluster *Cluster) StartVipService(ctx context.Context, c *kubevip.Config,
|
||||
backendMapV6 := backend.Map{}
|
||||
// only check localhost
|
||||
|
||||
ips := []string{}
|
||||
if c.NodeName != "" {
|
||||
if ips, err = getNodeIPs(ctx, c.NodeName, em.KubernetesClient); err != nil && !apierrors.IsNotFound(err) {
|
||||
log.Error("failed to get IP of control-plane node", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
if len(ips) == 0 {
|
||||
if !utils.IsIPv6(cluster.Network[0].IP()) {
|
||||
ips = append(ips, "127.0.0.1")
|
||||
} else {
|
||||
ips = append(ips, "::1")
|
||||
// An explicitly configured Kubernetes API address (static-pod
|
||||
// deployments point it at the local API server, whose loopback
|
||||
// listener is often the only certificate-valid local endpoint)
|
||||
// takes precedence over the Node object's addresses: the check
|
||||
// answers "is the local API server healthy" for every VIP family,
|
||||
// regardless of the transport family of the override itself.
|
||||
if entry := kubernetesAddrBackendEntry(c.KubernetesAddr, c.Port); entry != nil {
|
||||
log.Info("using configured Kubernetes address for backend health checks", "address", c.KubernetesAddr)
|
||||
backendMapV4[*entry] = false
|
||||
backendMapV6[*entry] = false
|
||||
} else {
|
||||
ips := []string{}
|
||||
if c.NodeName != "" {
|
||||
if ips, err = getNodeIPs(ctx, c.NodeName, em.KubernetesClient); err != nil && !apierrors.IsNotFound(err) {
|
||||
log.Error("failed to get IP of control-plane node", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
log.Info("no IP address found for node - will fallback to use localhost address", "addresses", ips)
|
||||
}
|
||||
if len(ips) == 0 {
|
||||
if !utils.IsIPv6(cluster.Network[0].IP()) {
|
||||
ips = append(ips, "127.0.0.1")
|
||||
} else {
|
||||
ips = append(ips, "::1")
|
||||
}
|
||||
|
||||
for _, ip := range ips {
|
||||
entry := backend.Entry{Addr: ip, Port: c.Port}
|
||||
if !utils.IsIPv6(ip) {
|
||||
backendMapV4[entry] = false
|
||||
} else {
|
||||
backendMapV6[entry] = false
|
||||
log.Info("no IP address found for node - will fallback to use localhost address", "addresses", ips)
|
||||
}
|
||||
|
||||
for _, ip := range ips {
|
||||
entry := backend.Entry{Addr: ip, Port: c.Port}
|
||||
if !utils.IsIPv6(ip) {
|
||||
backendMapV4[entry] = false
|
||||
} else {
|
||||
backendMapV6[entry] = false
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -273,7 +283,28 @@ func (cluster *Cluster) StartVipService(ctx context.Context, c *kubevip.Config,
|
||||
return nil
|
||||
}
|
||||
|
||||
func (cluster *Cluster) bgpHealthCheckLoop(ctx context.Context, c *kubevip.Config, bgpServer BGPRouteManager, vipCIDR string) {
|
||||
func (cluster *Cluster) bgpHealthCheck(ctx context.Context, c *kubevip.Config) (bool, error) {
|
||||
statusCode := 0
|
||||
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, c.ControlPlaneHealthCheck.Address, nil)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("building request %v: %w", req, err)
|
||||
} else {
|
||||
resp, err := cluster.healthCheckHTTPClient.Do(req)
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("checking control-plane: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
statusCode = resp.StatusCode
|
||||
}
|
||||
healthy := statusCode == http.StatusOK
|
||||
if !healthy {
|
||||
return healthy, fmt.Errorf("wrong status code: %d", statusCode)
|
||||
}
|
||||
return healthy, nil
|
||||
}
|
||||
|
||||
func (cluster *Cluster) bgpHealthCheckLoop(ctx context.Context, c *kubevip.Config, bgpServer bgp.BGPManager, vipCIDR string) {
|
||||
period := time.Duration(c.ControlPlaneHealthCheck.PeriodSeconds) * time.Second
|
||||
|
||||
consecutiveFailures := 0
|
||||
@@ -290,24 +321,7 @@ func (cluster *Cluster) bgpHealthCheckLoop(ctx context.Context, c *kubevip.Confi
|
||||
)
|
||||
|
||||
for {
|
||||
statusCode := 0
|
||||
var healthErr error
|
||||
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, c.ControlPlaneHealthCheck.Address, nil)
|
||||
if err != nil {
|
||||
healthErr = err
|
||||
} else {
|
||||
resp, err := cluster.healthCheckHTTPClient.Do(req)
|
||||
if err != nil {
|
||||
healthErr = err
|
||||
} else {
|
||||
defer resp.Body.Close()
|
||||
statusCode = resp.StatusCode
|
||||
}
|
||||
}
|
||||
|
||||
healthy := healthErr == nil && statusCode == http.StatusOK
|
||||
|
||||
healthy, healthErr := cluster.bgpHealthCheck(ctx, c)
|
||||
if healthy {
|
||||
consecutiveFailures = 0
|
||||
if !routeAnnounced {
|
||||
@@ -322,10 +336,7 @@ func (cluster *Cluster) bgpHealthCheckLoop(ctx context.Context, c *kubevip.Confi
|
||||
consecutiveFailures++
|
||||
if healthErr != nil {
|
||||
log.Warn("BGP health check failed", "address", c.ControlPlaneHealthCheck.Address, "consecutive", consecutiveFailures, "err", healthErr)
|
||||
} else {
|
||||
log.Warn("BGP health check failed", "address", c.ControlPlaneHealthCheck.Address, "consecutive", consecutiveFailures, "status", statusCode)
|
||||
}
|
||||
|
||||
if consecutiveFailures >= c.ControlPlaneHealthCheck.FailureThreshold && routeAnnounced {
|
||||
log.Warn("BGP health check threshold reached, withdrawing route", "failureThreshold", c.ControlPlaneHealthCheck.FailureThreshold, "cidr", vipCIDR)
|
||||
if err := bgpServer.DelHost(ctx, vipCIDR, c.NodeName); err != nil {
|
||||
@@ -349,6 +360,27 @@ func (cluster *Cluster) bgpHealthCheckLoop(ctx context.Context, c *kubevip.Confi
|
||||
}
|
||||
}
|
||||
|
||||
// kubernetesAddrBackendEntry converts an explicitly configured Kubernetes
|
||||
// API address override (config.KubernetesAddr, e.g. "https://127.0.0.1:6443"
|
||||
// on static-pod deployments) into a backend health-check entry. Returns nil
|
||||
// when no usable override is configured.
|
||||
func kubernetesAddrBackendEntry(kubernetesAddr string, defaultPort uint16) *backend.Entry {
|
||||
if kubernetesAddr == "" {
|
||||
return nil
|
||||
}
|
||||
u, err := url.Parse(kubernetesAddr)
|
||||
if err != nil || u.Hostname() == "" {
|
||||
return nil
|
||||
}
|
||||
port := defaultPort
|
||||
if p := u.Port(); p != "" {
|
||||
if parsed, err := strconv.ParseUint(p, 10, 16); err == nil {
|
||||
port = uint16(parsed)
|
||||
}
|
||||
}
|
||||
return &backend.Entry{Addr: u.Hostname(), Port: port}
|
||||
}
|
||||
|
||||
func getNodeIPs(ctx context.Context, nodename string, client *kubernetes.Clientset) ([]string, error) {
|
||||
node, err := client.CoreV1().Nodes().Get(ctx, nodename, metav1.GetOptions{})
|
||||
if err != nil && !apierrors.IsNotFound(err) {
|
||||
@@ -364,16 +396,59 @@ func getNodeIPs(ctx context.Context, nodename string, client *kubernetes.Clients
|
||||
}
|
||||
|
||||
// StartLoadBalancerService will start a VIP instance and leave it for kube-proxy to handle
|
||||
func (cluster *Cluster) StartLoadBalancerService(ctx context.Context, c *kubevip.Config, bgp *bgp.Server, name string, wg *sync.WaitGroup) error {
|
||||
func (cluster *Cluster) StartLoadBalancerService(ctx context.Context, c *kubevip.Config, bgp bgp.BGPManager, name string, wg *sync.WaitGroup) error {
|
||||
// use a Go context so we can tell the arp loop code when we
|
||||
// want to step down
|
||||
//nolint
|
||||
lbCtx, lbCancel := context.WithCancel(ctx)
|
||||
|
||||
var lbWg sync.WaitGroup
|
||||
stop, done, err := cluster.startServicesWorker()
|
||||
if err != nil {
|
||||
lbCancel()
|
||||
return err
|
||||
}
|
||||
type startedNetwork struct {
|
||||
network vip.Network
|
||||
routeAdded bool
|
||||
ipAdded bool
|
||||
bgpAdded bool
|
||||
}
|
||||
startedNetworks := make([]startedNetwork, 0, len(cluster.Network))
|
||||
servicesWorkerStarted := false
|
||||
defer func() {
|
||||
if !servicesWorkerStarted {
|
||||
lbCancel()
|
||||
lbWg.Wait()
|
||||
cleanupCtx := context.WithoutCancel(ctx)
|
||||
for index := len(startedNetworks) - 1; index >= 0; index-- {
|
||||
started := startedNetworks[index]
|
||||
if started.bgpAdded && bgp != nil {
|
||||
if err := bgp.DelHost(cleanupCtx, started.network.CIDR(), name); err != nil {
|
||||
log.Warn("failed to withdraw BGP host after startup failure", "address", started.network.CIDR(), "err", err)
|
||||
}
|
||||
}
|
||||
if started.routeAdded && cluster.routeMgr != nil {
|
||||
if err := cluster.routeMgr.Delete(name, started.network); err != nil {
|
||||
log.Warn("failed to delete route after startup failure", "address", started.network.CIDR(), "err", err)
|
||||
}
|
||||
}
|
||||
if started.ipAdded {
|
||||
if _, err := started.network.DeleteIP(); err != nil {
|
||||
log.Warn("failed to delete VIP after startup failure", "address", started.network.IP(), "err", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
cluster.finishServicesWorker(done)
|
||||
}
|
||||
}()
|
||||
|
||||
for i := range cluster.Network {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
network := cluster.Network[i]
|
||||
startedNetworks = append(startedNetworks, startedNetwork{network: network})
|
||||
started := &startedNetworks[len(startedNetworks)-1]
|
||||
|
||||
if network.IsDDNS() {
|
||||
ddnsReady := make(chan struct{})
|
||||
@@ -394,11 +469,12 @@ func (cluster *Cluster) StartLoadBalancerService(ctx context.Context, c *kubevip
|
||||
if err := network.SetMask(c.VIPSubnet); err != nil {
|
||||
log.Error("failed to set mask", "subnet", c.VIPSubnet, "err", err)
|
||||
lbCancel()
|
||||
return utils.NewPanicError(fmt.Sprintf("failed to set mask for subnet %q: %s", c.VIPSubnet, err.Error()))
|
||||
return utils.WrapPanicError(err, "failed to set mask for subnet %q", c.VIPSubnet)
|
||||
}
|
||||
_, err := network.DeleteIP()
|
||||
existing, err := network.IsSet()
|
||||
if err != nil {
|
||||
log.Warn("attempted to clean existing VIP", "err", err)
|
||||
lbCancel()
|
||||
return fmt.Errorf("check existing VIP %q: %w", network.IP(), err)
|
||||
}
|
||||
log.Debug("config flags", "enable_routing_table", c.EnableRoutingTable, "enable_leader_election", c.EnableLeaderElection, "enable_services_election", c.EnableServicesElection)
|
||||
|
||||
@@ -407,16 +483,19 @@ func (cluster *Cluster) StartLoadBalancerService(ctx context.Context, c *kubevip
|
||||
if err != nil {
|
||||
log.Warn(err.Error())
|
||||
} else {
|
||||
started.routeAdded = true
|
||||
log.Info("successful add Route")
|
||||
}
|
||||
}
|
||||
|
||||
if !c.EnableRoutingTable && !c.EnableBGP && !c.EnableWireguard {
|
||||
if shouldAddServiceIP(c) {
|
||||
// Normal VIP addition, use skipDAD=false for normal DAD process
|
||||
// Note: When WireGuard is enabled, the VIP is added to the tunnel interface
|
||||
// instead of lo, so we skip adding it here.
|
||||
if _, err = network.AddIP(false, false); err != nil {
|
||||
log.Warn(err.Error())
|
||||
added, addErr := network.AddIP(false, false)
|
||||
started.ipAdded = existing == nil && added
|
||||
if addErr != nil {
|
||||
log.Warn(addErr.Error())
|
||||
} else {
|
||||
log.Info("successful add IP", "address", network.IP())
|
||||
}
|
||||
@@ -434,11 +513,14 @@ func (cluster *Cluster) StartLoadBalancerService(ctx context.Context, c *kubevip
|
||||
err = bgp.AddHost(lbCtx, network.CIDR(), name)
|
||||
if err != nil {
|
||||
log.Error(err.Error())
|
||||
} else {
|
||||
started.bgpAdded = true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
wg.Go(func() {
|
||||
defer cluster.finishServicesWorker(done)
|
||||
for i := range cluster.Network {
|
||||
network := cluster.Network[i]
|
||||
|
||||
@@ -453,7 +535,7 @@ func (cluster *Cluster) StartLoadBalancerService(ctx context.Context, c *kubevip
|
||||
}
|
||||
|
||||
select {
|
||||
case <-cluster.stop:
|
||||
case <-stop:
|
||||
case <-ctx.Done():
|
||||
}
|
||||
|
||||
@@ -474,12 +556,17 @@ func (cluster *Cluster) StartLoadBalancerService(ctx context.Context, c *kubevip
|
||||
return
|
||||
}
|
||||
|
||||
cluster.cleanupVIPs(c)
|
||||
cluster.cleanupServiceVIPs(c, done)
|
||||
})
|
||||
servicesWorkerStarted = true
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func shouldAddServiceIP(c *kubevip.Config) bool {
|
||||
return !c.EnableRoutingTable && (!c.EnableBGP || c.BGPAttachIPToInterface) && !c.EnableWireguard
|
||||
}
|
||||
|
||||
// Layer2Update, handles the creation of the
|
||||
func (cluster *Cluster) layer2Update(ctx context.Context, network vip.Network, c *kubevip.Config) {
|
||||
var ndp *vip.NdpResponder
|
||||
|
||||
55
pkg/cluster/service_config_test.go
Normal file
55
pkg/cluster/service_config_test.go
Normal file
@@ -0,0 +1,55 @@
|
||||
package cluster
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
)
|
||||
|
||||
func TestShouldAddServiceIP(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
config *kubevip.Config
|
||||
want bool
|
||||
}{
|
||||
{
|
||||
name: "BGP default does not attach IP",
|
||||
config: &kubevip.Config{EnableBGP: true},
|
||||
want: false,
|
||||
},
|
||||
{
|
||||
name: "BGP opt-in attaches IP",
|
||||
config: &kubevip.Config{
|
||||
EnableBGP: true,
|
||||
BGPAttachIPToInterface: true,
|
||||
},
|
||||
want: true,
|
||||
},
|
||||
{
|
||||
name: "routing table takes precedence",
|
||||
config: &kubevip.Config{
|
||||
EnableBGP: true,
|
||||
BGPAttachIPToInterface: true,
|
||||
EnableRoutingTable: true,
|
||||
},
|
||||
want: false,
|
||||
},
|
||||
{
|
||||
name: "WireGuard takes precedence",
|
||||
config: &kubevip.Config{
|
||||
EnableBGP: true,
|
||||
BGPAttachIPToInterface: true,
|
||||
EnableWireguard: true,
|
||||
},
|
||||
want: false,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if got := shouldAddServiceIP(tt.config); got != tt.want {
|
||||
t.Fatalf("shouldAddServiceIP() = %t, want %t", got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
61
pkg/cluster/service_internal_test.go
Normal file
61
pkg/cluster/service_internal_test.go
Normal file
@@ -0,0 +1,61 @@
|
||||
package cluster
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestKubernetesAddrBackendEntry(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
addr string
|
||||
port uint16
|
||||
wantAddr string
|
||||
wantPort uint16
|
||||
wantNil bool
|
||||
}{
|
||||
{
|
||||
name: "explicit v4 loopback with port",
|
||||
addr: "https://127.0.0.1:6443",
|
||||
port: 9999,
|
||||
wantAddr: "127.0.0.1",
|
||||
wantPort: 6443,
|
||||
},
|
||||
{
|
||||
name: "hostname without port falls back to config port",
|
||||
addr: "https://localhost",
|
||||
port: 6443,
|
||||
wantAddr: "localhost",
|
||||
wantPort: 6443,
|
||||
},
|
||||
{
|
||||
name: "empty override",
|
||||
addr: "",
|
||||
port: 6443,
|
||||
wantNil: true,
|
||||
},
|
||||
{
|
||||
name: "garbage override",
|
||||
addr: "://not-a-url",
|
||||
port: 6443,
|
||||
wantNil: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
entry := kubernetesAddrBackendEntry(tc.addr, tc.port)
|
||||
if tc.wantNil {
|
||||
if entry != nil {
|
||||
t.Fatalf("expected nil entry, got %+v", entry)
|
||||
}
|
||||
return
|
||||
}
|
||||
if entry == nil {
|
||||
t.Fatal("expected an entry, got nil")
|
||||
}
|
||||
if entry.Addr != tc.wantAddr || entry.Port != tc.wantPort {
|
||||
t.Fatalf("got %+v, want addr %q port %d", entry, tc.wantAddr, tc.wantPort)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,7 @@ package cluster_test
|
||||
import (
|
||||
"context"
|
||||
"encoding/pem"
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
@@ -12,6 +13,7 @@ import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/bgp"
|
||||
"github.com/kube-vip/kube-vip/pkg/cluster"
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/route"
|
||||
@@ -28,19 +30,123 @@ func TestBGPHealthCheckLoop_AnnouncesOnHealthy(t *testing.T) {
|
||||
t.Cleanup(healthcheck.server.Close)
|
||||
|
||||
bgpManager := newMockBGPRouteManager()
|
||||
startVipService(t, newTestConfig(healthcheck.server.URL, healthcheck.caPath), bgpManager)
|
||||
startVipService(t, newBGPConfig(healthcheck.server.URL, healthcheck.caPath), bgpManager)
|
||||
|
||||
expectEventually(t, func() bool { return bgpManager.isAnnounced() },
|
||||
"route should be announced")
|
||||
}
|
||||
|
||||
func TestServicesWorkerStopAndWaitDrainsBeforeRestart(t *testing.T) {
|
||||
config := &kubevip.Config{}
|
||||
serviceCluster, err := cluster.InitCluster(config, true, nil, nil, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("InitCluster() error = %v", err)
|
||||
}
|
||||
var workers sync.WaitGroup
|
||||
if err := serviceCluster.StartLoadBalancerService(context.Background(), config, nil, "service", &workers); err != nil {
|
||||
t.Fatalf("first StartLoadBalancerService() error = %v", err)
|
||||
}
|
||||
if err := serviceCluster.StartLoadBalancerService(context.Background(), config, nil, "service", &workers); err == nil {
|
||||
t.Fatal("second StartLoadBalancerService() started while the first workers were active")
|
||||
}
|
||||
|
||||
serviceCluster.StopAndWait()
|
||||
if err := serviceCluster.StartLoadBalancerService(context.Background(), config, nil, "service", &workers); err != nil {
|
||||
t.Fatalf("StartLoadBalancerService() after StopAndWait error = %v", err)
|
||||
}
|
||||
serviceCluster.StopAndWait()
|
||||
workers.Wait()
|
||||
}
|
||||
|
||||
func TestServicesWorkerStopAndWaitPreservingDrainsBeforeRestart(t *testing.T) {
|
||||
config := &kubevip.Config{}
|
||||
serviceCluster, err := cluster.InitCluster(config, true, nil, nil, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("InitCluster() error = %v", err)
|
||||
}
|
||||
var workers sync.WaitGroup
|
||||
if err := serviceCluster.StartLoadBalancerService(context.Background(), config, nil, "service", &workers); err != nil {
|
||||
t.Fatalf("StartLoadBalancerService() error = %v", err)
|
||||
}
|
||||
serviceCluster.StopAndWait()
|
||||
if err := serviceCluster.StartLoadBalancerService(context.Background(), config, nil, "service", &workers); err != nil {
|
||||
t.Fatalf("StartLoadBalancerService() after preserving stop error = %v", err)
|
||||
}
|
||||
serviceCluster.StopAndWait()
|
||||
workers.Wait()
|
||||
}
|
||||
|
||||
func TestStartLoadBalancerServiceRollsBackEarlierNetwork(t *testing.T) {
|
||||
first := &mockNetwork{ip: "192.0.2.10", cidr: "192.0.2.10/32"}
|
||||
second := &mockNetwork{ip: "192.0.2.11", cidr: "192.0.2.11/32", setMaskErr: errors.New("set mask")}
|
||||
serviceCluster, err := cluster.InitCluster(&kubevip.Config{}, true, nil, nil, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("InitCluster() error = %v", err)
|
||||
}
|
||||
serviceCluster.Network = []vip.Network{first, second}
|
||||
|
||||
if err := serviceCluster.StartLoadBalancerService(context.Background(), &kubevip.Config{VIPSubnet: "32"}, nil, "service", &sync.WaitGroup{}); err == nil {
|
||||
t.Fatal("StartLoadBalancerService() error = nil, want second-network failure")
|
||||
}
|
||||
first.mu.Lock()
|
||||
addCalls, deleteCalls, present := first.addIPCalls, first.deleteIPCalls, first.present
|
||||
first.mu.Unlock()
|
||||
if addCalls != 1 || deleteCalls != 1 || present {
|
||||
t.Fatalf("first network rollback = add %d, delete %d, present %t; want 1, 1, false", addCalls, deleteCalls, present)
|
||||
}
|
||||
serviceCluster.StopAndWait()
|
||||
}
|
||||
|
||||
func TestStartLoadBalancerServiceRollbackPreservesExistingVIP(t *testing.T) {
|
||||
first := &mockNetwork{ip: "192.0.2.10", cidr: "192.0.2.10/32", present: true}
|
||||
second := &mockNetwork{ip: "192.0.2.11", cidr: "192.0.2.11/32", setMaskErr: errors.New("set mask")}
|
||||
serviceCluster, err := cluster.InitCluster(&kubevip.Config{}, true, nil, nil, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("InitCluster() error = %v", err)
|
||||
}
|
||||
serviceCluster.Network = []vip.Network{first, second}
|
||||
|
||||
if err := serviceCluster.StartLoadBalancerService(context.Background(), &kubevip.Config{VIPSubnet: "32"}, nil, "service", &sync.WaitGroup{}); err == nil {
|
||||
t.Fatal("StartLoadBalancerService() error = nil, want second-network failure")
|
||||
}
|
||||
first.mu.Lock()
|
||||
addCalls, deleteCalls, present := first.addIPCalls, first.deleteIPCalls, first.present
|
||||
first.mu.Unlock()
|
||||
if addCalls != 1 || deleteCalls != 0 || !present {
|
||||
t.Fatalf("existing VIP rollback = add %d, delete %d, present %t; want 1, 0, true", addCalls, deleteCalls, present)
|
||||
}
|
||||
serviceCluster.StopAndWait()
|
||||
}
|
||||
|
||||
func TestStartLoadBalancerServiceCancelledContextDoesNotConfigureVIP(t *testing.T) {
|
||||
network := &mockNetwork{ip: "192.0.2.10", cidr: "192.0.2.10/32"}
|
||||
serviceCluster, err := cluster.InitCluster(&kubevip.Config{}, true, nil, nil, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("InitCluster() error = %v", err)
|
||||
}
|
||||
serviceCluster.Network = []vip.Network{network}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
|
||||
err = serviceCluster.StartLoadBalancerService(ctx, &kubevip.Config{VIPSubnet: "32"}, nil, "service", &sync.WaitGroup{})
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("StartLoadBalancerService() error = %v, want context.Canceled", err)
|
||||
}
|
||||
network.mu.Lock()
|
||||
addCalls := network.addIPCalls
|
||||
network.mu.Unlock()
|
||||
if addCalls != 0 {
|
||||
t.Fatalf("AddIP calls = %d, want 0 after context cancellation", addCalls)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBGPHealthCheckLoop_NoAnnouncementUntilHealthy(t *testing.T) {
|
||||
t.Parallel()
|
||||
healthcheck := newTestHealthServer(t, http.StatusInternalServerError)
|
||||
t.Cleanup(healthcheck.server.Close)
|
||||
|
||||
bgpManager := newMockBGPRouteManager()
|
||||
startVipService(t, newTestConfig(healthcheck.server.URL, healthcheck.caPath), bgpManager)
|
||||
startVipService(t, newBGPConfig(healthcheck.server.URL, healthcheck.caPath), bgpManager)
|
||||
|
||||
expectConsistently(t, func() bool { return !bgpManager.isAnnounced() },
|
||||
2*time.Second, "route should not be announced while unhealthy")
|
||||
@@ -56,7 +162,7 @@ func TestBGPHealthCheckLoop_WithdrawsAfterThreshold(t *testing.T) {
|
||||
t.Cleanup(healthcheck.server.Close)
|
||||
|
||||
bgpManager := newMockBGPRouteManager()
|
||||
cfg := newTestConfig(healthcheck.server.URL, healthcheck.caPath)
|
||||
cfg := newBGPConfig(healthcheck.server.URL, healthcheck.caPath)
|
||||
cfg.ControlPlaneHealthCheck.FailureThreshold = 3
|
||||
startVipService(t, cfg, bgpManager)
|
||||
|
||||
@@ -78,7 +184,7 @@ func TestBGPHealthCheckLoop_ReAnnouncesOnRecovery(t *testing.T) {
|
||||
t.Cleanup(healthcheck.server.Close)
|
||||
|
||||
bgpManager := newMockBGPRouteManager()
|
||||
cfg := newTestConfig(healthcheck.server.URL, healthcheck.caPath)
|
||||
cfg := newBGPConfig(healthcheck.server.URL, healthcheck.caPath)
|
||||
cfg.ControlPlaneHealthCheck.FailureThreshold = 1
|
||||
startVipService(t, cfg, bgpManager)
|
||||
|
||||
@@ -100,7 +206,7 @@ func TestBGPHealthCheckLoop_StopsOnContextCancel(t *testing.T) {
|
||||
t.Cleanup(healthcheck.server.Close)
|
||||
|
||||
bgpManager := newMockBGPRouteManager()
|
||||
cancelContext, vipServiceDone := startVipService(t, newTestConfig(healthcheck.server.URL, healthcheck.caPath), bgpManager)
|
||||
cancelContext, vipServiceDone := startVipService(t, newBGPConfig(healthcheck.server.URL, healthcheck.caPath), bgpManager)
|
||||
|
||||
expectEventually(t, func() bool { return bgpManager.isAnnounced() },
|
||||
"route should be announced")
|
||||
@@ -121,7 +227,7 @@ func TestBGPHealthCheckLoop_RetriesAddHostOnFailure(t *testing.T) {
|
||||
|
||||
bgpManager := newMockBGPRouteManager()
|
||||
bgpManager.setAddErr(errTestAddHost)
|
||||
startVipService(t, newTestConfig(healthcheck.server.URL, healthcheck.caPath), bgpManager)
|
||||
startVipService(t, newBGPConfig(healthcheck.server.URL, healthcheck.caPath), bgpManager)
|
||||
|
||||
expectConsistently(t, func() bool { return !bgpManager.isAnnounced() },
|
||||
2*time.Second, "route should not be announced while AddHost errors")
|
||||
@@ -137,7 +243,7 @@ func TestBGPHealthCheckLoop_RetriesDelHostOnFailure(t *testing.T) {
|
||||
t.Cleanup(healthcheck.server.Close)
|
||||
|
||||
bgpManager := newMockBGPRouteManager()
|
||||
cfg := newTestConfig(healthcheck.server.URL, healthcheck.caPath)
|
||||
cfg := newBGPConfig(healthcheck.server.URL, healthcheck.caPath)
|
||||
cfg.ControlPlaneHealthCheck.FailureThreshold = 1
|
||||
startVipService(t, cfg, bgpManager)
|
||||
|
||||
@@ -195,9 +301,8 @@ func (e *testError) Error() string { return e.msg }
|
||||
// startVipService launches vipService in a goroutine with a mock network and
|
||||
// registers a cleanup to cancel the context and wait for it to finish.
|
||||
// Uses InitCluster so the real code parses certs for the BGP health check client.
|
||||
func startVipService(t *testing.T, cfg *kubevip.Config, bgpManager *mockBGPRouteManager) (context.CancelFunc, <-chan struct{}) {
|
||||
func startVipService(t *testing.T, cfg *kubevip.Config, bgpServer bgp.BGPManager) (context.CancelFunc, <-chan struct{}) {
|
||||
t.Helper()
|
||||
|
||||
c, err := cluster.InitCluster(cfg, true, nil, nil, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("InitCluster: %v", err)
|
||||
@@ -208,7 +313,7 @@ func startVipService(t *testing.T, cfg *kubevip.Config, bgpManager *mockBGPRoute
|
||||
done := make(chan struct{})
|
||||
|
||||
go func() {
|
||||
_ = c.StartVipService(ctx, cfg, nil, bgpManager, func() {})
|
||||
_ = c.StartVipService(ctx, cfg, nil, bgpServer, func() {})
|
||||
close(done)
|
||||
}()
|
||||
|
||||
@@ -247,14 +352,14 @@ func startRoutingTableVipService(t *testing.T, cfg *kubevip.Config, network *moc
|
||||
}
|
||||
|
||||
func newRoutingTableConfig(url, caPath string) *kubevip.Config {
|
||||
cfg := newTestConfig(url, caPath)
|
||||
cfg := newBGPConfig(url, caPath)
|
||||
cfg.EnableBGP = false
|
||||
cfg.EnableRoutingTable = true
|
||||
cfg.BackendHealthCheckInterval = 1
|
||||
return cfg
|
||||
}
|
||||
|
||||
func newTestConfig(url, caPath string) *kubevip.Config {
|
||||
func newBGPConfig(url, caPath string) *kubevip.Config {
|
||||
return &kubevip.Config{
|
||||
EnableBGP: true,
|
||||
ControlPlaneHealthCheck: kubevip.HealthCheck{
|
||||
@@ -323,32 +428,45 @@ type mockNetwork struct {
|
||||
ip string
|
||||
cidr string
|
||||
|
||||
mu sync.Mutex
|
||||
present bool
|
||||
mu sync.Mutex
|
||||
present bool
|
||||
setMaskErr error
|
||||
addIPCalls int
|
||||
deleteIPCalls int
|
||||
}
|
||||
|
||||
func (m *mockNetwork) AddIP(bool, bool, ...int) (bool, error) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.addIPCalls++
|
||||
m.present = true
|
||||
return true, nil
|
||||
}
|
||||
func (m *mockNetwork) DeleteIP() (bool, error) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.deleteIPCalls++
|
||||
deleted := m.present
|
||||
m.present = false
|
||||
return false, nil
|
||||
return deleted, nil
|
||||
}
|
||||
func (m *mockNetwork) isPresent() bool {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
return m.present
|
||||
}
|
||||
func (m *mockNetwork) AddRoute(bool) (bool, error) { return false, nil }
|
||||
func (m *mockNetwork) ReplaceRoute() error { return nil }
|
||||
func (m *mockNetwork) DeleteRoute() error { return nil }
|
||||
func (m *mockNetwork) UpdateRoutes() (bool, error) { return false, nil }
|
||||
func (m *mockNetwork) IsSet() (*netlink.Addr, error) { return nil, nil }
|
||||
func (m *mockNetwork) AddRoute(bool) (bool, error) { return false, nil }
|
||||
func (m *mockNetwork) ReplaceRoute() error { return nil }
|
||||
func (m *mockNetwork) DeleteRoute() error { return nil }
|
||||
func (m *mockNetwork) UpdateRoutes() (bool, error) { return false, nil }
|
||||
func (m *mockNetwork) IsSet() (*netlink.Addr, error) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if m.present {
|
||||
return &netlink.Addr{}, nil
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
func (m *mockNetwork) IP() string { return m.ip }
|
||||
func (m *mockNetwork) CIDR() string { return m.cidr }
|
||||
func (m *mockNetwork) IPisLinkLocal() bool { return false }
|
||||
@@ -362,7 +480,7 @@ func (m *mockNetwork) IsDNS() bool { return false }
|
||||
func (m *mockNetwork) IsDDNS() bool { return false }
|
||||
func (m *mockNetwork) DDNSHostName() string { return "" }
|
||||
func (m *mockNetwork) DNSName() string { return "" }
|
||||
func (m *mockNetwork) SetMask(string) error { return nil }
|
||||
func (m *mockNetwork) SetMask(string) error { return m.setMaskErr }
|
||||
func (m *mockNetwork) SetHasEndpoints(bool) {}
|
||||
func (m *mockNetwork) HasEndpoints() bool { return false }
|
||||
func (m *mockNetwork) ARPName() string { return "" }
|
||||
|
||||
@@ -52,9 +52,8 @@ func (n *ns) add(name string, output chan<- watch.Event) *object {
|
||||
return i
|
||||
}
|
||||
|
||||
func (n *ns) del(name string) {
|
||||
if _, exists := n.Load(name); exists {
|
||||
n.Delete(name)
|
||||
func (n *ns) del(name string, object *object) {
|
||||
if n.CompareAndDelete(name, object) {
|
||||
n.cnt.Add(-1)
|
||||
}
|
||||
}
|
||||
@@ -119,35 +118,57 @@ func (d *debouncer) Start(ctx context.Context) error {
|
||||
return fmt.Errorf("objects of type %T are not supported", v)
|
||||
}
|
||||
|
||||
eventNs, exists := d.getNs(namespace)
|
||||
if !exists {
|
||||
// if not, create new map for the namespace
|
||||
eventNs = d.addNs(namespace)
|
||||
}
|
||||
processEvent:
|
||||
for {
|
||||
eventNs, exists := d.getNs(namespace)
|
||||
if !exists {
|
||||
// if not, create new map for the namespace
|
||||
eventNs = d.addNs(namespace)
|
||||
}
|
||||
|
||||
// check if the object was previously reconciled
|
||||
eventObject, exists := eventNs.get(name)
|
||||
// check if the object was previously reconciled
|
||||
eventObject, exists := eventNs.get(name)
|
||||
|
||||
// if not and the event is not of type 'Deleted', create new object
|
||||
if !exists && tmp.Type != watch.Deleted {
|
||||
eventObject = eventNs.add(name, d.output)
|
||||
// if not and the event is not of type 'Deleted', create new object
|
||||
if !exists && tmp.Type != watch.Deleted {
|
||||
eventObject = eventNs.add(name, d.output)
|
||||
|
||||
wg.Go(func() {
|
||||
// start deboucing events for this object
|
||||
eventObject.start(debouncerCtx, d.debounceTime)
|
||||
// if debouncer for the object ended - e.g. object was deleted - clean the map of objects
|
||||
eventObject = nil
|
||||
eventNs.del(name)
|
||||
// if namespace is empty, delete the namespace map
|
||||
if eventNs.cnt.Load() == 0 {
|
||||
d.delNs(namespace)
|
||||
workerObject := eventObject
|
||||
workerNs := eventNs
|
||||
workerName := name
|
||||
workerNamespace := namespace
|
||||
workerObject.onStop = func() {
|
||||
// Remove the object before its worker can become receiver-less.
|
||||
workerNs.del(workerName, workerObject)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
if eventObject != nil {
|
||||
wg.Go(func() {
|
||||
// start deboucing events for this object
|
||||
workerObject.start(debouncerCtx, d.debounceTime)
|
||||
// if debouncer for the object ended - e.g. object was deleted - clean the map of objects
|
||||
workerNs.del(workerName, workerObject)
|
||||
// if namespace is empty, delete the namespace map
|
||||
if workerNs.cnt.Load() == 0 {
|
||||
d.delNs(workerNamespace, workerNs)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
if eventObject == nil {
|
||||
break processEvent
|
||||
}
|
||||
|
||||
// pass the watch event to the debouncer object
|
||||
eventObject.input <- tmp
|
||||
select {
|
||||
case eventObject.input <- tmp:
|
||||
break processEvent
|
||||
case <-eventObject.stopChan:
|
||||
// The object stopped after the map lookup. Retry the event
|
||||
// against the newly-created object instead of dropping it.
|
||||
continue processEvent
|
||||
case <-debouncerCtx.Done():
|
||||
return nil
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -181,8 +202,8 @@ func (d *debouncer) addNs(namespace string) *ns {
|
||||
return &n
|
||||
}
|
||||
|
||||
func (d *debouncer) delNs(namespace string) {
|
||||
d.namespaces.Delete(namespace)
|
||||
func (d *debouncer) delNs(namespace string, ns *ns) {
|
||||
d.namespaces.CompareAndDelete(namespace, ns)
|
||||
}
|
||||
|
||||
type object struct {
|
||||
@@ -190,6 +211,7 @@ type object struct {
|
||||
output chan<- watch.Event
|
||||
stopChan chan any
|
||||
stopOnce sync.Once
|
||||
onStop func()
|
||||
}
|
||||
|
||||
func newObject(output chan<- watch.Event) *object {
|
||||
@@ -247,6 +269,9 @@ func (o *object) start(ctx context.Context, debounceTime time.Duration) {
|
||||
|
||||
func (o *object) stop() {
|
||||
o.stopOnce.Do(func() {
|
||||
if o.onStop != nil {
|
||||
o.onStop()
|
||||
}
|
||||
close(o.stopChan)
|
||||
})
|
||||
}
|
||||
|
||||
162
pkg/debouncer/debouncer_deadlock_test.go
Normal file
162
pkg/debouncer/debouncer_deadlock_test.go
Normal file
@@ -0,0 +1,162 @@
|
||||
package debouncer
|
||||
|
||||
import (
|
||||
"context"
|
||||
"runtime"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
v1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/watch"
|
||||
)
|
||||
|
||||
func TestStartReturnsWhenCancellationInterruptsObjectForwarding(t *testing.T) {
|
||||
input := make(chan watch.Event)
|
||||
d := &debouncer{
|
||||
input: input,
|
||||
output: make(chan watch.Event),
|
||||
stopChan: make(chan any),
|
||||
debounceTime: 200 * time.Millisecond,
|
||||
}
|
||||
|
||||
// Leave the object without a receiver. This is the state reached when its
|
||||
// worker exits on context cancellation just before Start forwards an event.
|
||||
ns := d.addNs("default")
|
||||
ns.add("example", d.output)
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- d.Start(ctx) }()
|
||||
|
||||
event := watch.Event{
|
||||
Type: watch.Modified,
|
||||
Object: &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "example", Namespace: "default",
|
||||
}},
|
||||
}
|
||||
sent := make(chan struct{})
|
||||
go func() {
|
||||
input <- event
|
||||
close(sent)
|
||||
}()
|
||||
select {
|
||||
case <-sent:
|
||||
case <-time.After(250 * time.Millisecond):
|
||||
cancel()
|
||||
t.Fatal("debouncer did not receive the test event")
|
||||
}
|
||||
cancel()
|
||||
|
||||
// Nothing ever receives from the object, so Start can only return by
|
||||
// abandoning the blocked forward when the context is cancelled. Receiving
|
||||
// here instead would make the forward succeed and the assertion racy.
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("debouncer remained blocked forwarding an event after context cancellation")
|
||||
}
|
||||
}
|
||||
|
||||
func TestStartRecreatesObjectAfterDeletionWithoutCancellation(t *testing.T) {
|
||||
previousProcs := runtime.GOMAXPROCS(1)
|
||||
t.Cleanup(func() { runtime.GOMAXPROCS(previousProcs) })
|
||||
|
||||
input := make(chan watch.Event)
|
||||
d, err := New(input, "200ms")
|
||||
if err != nil {
|
||||
t.Fatalf("failed to create debouncer: %s", err)
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
t.Cleanup(cancel)
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- d.Start(ctx) }()
|
||||
|
||||
service := func(eventType watch.EventType, resourceVersion string) watch.Event {
|
||||
return watch.Event{
|
||||
Type: eventType,
|
||||
Object: &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "example",
|
||||
Namespace: "default",
|
||||
ResourceVersion: resourceVersion,
|
||||
}},
|
||||
}
|
||||
}
|
||||
|
||||
send := func(event watch.Event) {
|
||||
t.Helper()
|
||||
select {
|
||||
case input <- event:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("debouncer did not receive the test event")
|
||||
}
|
||||
}
|
||||
|
||||
send(service(watch.Added, "initial"))
|
||||
select {
|
||||
case event := <-d.output:
|
||||
if event.Type != watch.Added {
|
||||
t.Fatalf("expected initial Added event, got %s", event.Type)
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("debouncer did not emit the initial event")
|
||||
}
|
||||
|
||||
eventNs, exists := d.getNs("default")
|
||||
if !exists {
|
||||
t.Fatal("debouncer did not create the namespace map")
|
||||
}
|
||||
oldObject, exists := eventNs.get("example")
|
||||
if !exists {
|
||||
t.Fatal("debouncer did not create the object")
|
||||
}
|
||||
|
||||
send(service(watch.Deleted, "deleted"))
|
||||
select {
|
||||
case event := <-d.output:
|
||||
if event.Type != watch.Deleted {
|
||||
t.Fatalf("expected Deleted event, got %s", event.Type)
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("debouncer did not emit the Deleted event")
|
||||
}
|
||||
|
||||
select {
|
||||
case <-oldObject.stopChan:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("object did not self-terminate")
|
||||
}
|
||||
if _, exists := eventNs.get("example"); exists {
|
||||
t.Fatal("self-terminated object remained in the namespace map")
|
||||
}
|
||||
|
||||
send(service(watch.Modified, "fresh"))
|
||||
select {
|
||||
case event := <-d.output:
|
||||
if event.Type != watch.Modified {
|
||||
t.Fatalf("expected fresh Modified event, got %s", event.Type)
|
||||
}
|
||||
service, ok := event.Object.(*v1.Service)
|
||||
if !ok {
|
||||
t.Fatalf("expected a Service event, got %T", event.Object)
|
||||
}
|
||||
if service.ResourceVersion != "fresh" {
|
||||
t.Fatalf("expected the fresh event, got resource version %q", service.ResourceVersion)
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("debouncer did not process the fresh event after object deletion")
|
||||
}
|
||||
|
||||
cancel()
|
||||
select {
|
||||
case err := <-done:
|
||||
if err != nil {
|
||||
t.Fatalf("debouncer returned an error: %s", err)
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("debouncer did not stop")
|
||||
}
|
||||
}
|
||||
@@ -2,14 +2,12 @@ package election
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
log "log/slog"
|
||||
|
||||
"github.com/davecgh/go-spew/spew"
|
||||
"github.com/kube-vip/kube-vip/pkg/etcd"
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
@@ -17,7 +15,6 @@ import (
|
||||
"github.com/kube-vip/kube-vip/pkg/utils"
|
||||
clientv3 "go.etcd.io/etcd/client/v3"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/watch"
|
||||
"k8s.io/client-go/kubernetes"
|
||||
@@ -62,7 +59,7 @@ func NewManager(config *kubevip.Config, k8sClientset, rwClientset *kubernetes.Cl
|
||||
func RunOrDie(ctx context.Context, run *RunConfig, c *kubevip.Config) error {
|
||||
switch c.LeaderElectionType {
|
||||
case "kubernetes", "":
|
||||
runKubernetesLeaderElectionOrDie(ctx, run)
|
||||
return runKubernetesLeaderElectionOrDie(ctx, run)
|
||||
case "etcd":
|
||||
if err := runEtcdLeaderElectionOrDie(ctx, run); err != nil {
|
||||
return err
|
||||
@@ -74,20 +71,25 @@ func RunOrDie(ctx context.Context, run *RunConfig, c *kubevip.Config) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
func runKubernetesLeaderElectionOrDie(ctx context.Context, run *RunConfig) {
|
||||
func runKubernetesLeaderElectionOrDie(ctx context.Context, run *RunConfig) error {
|
||||
annotations, err := kubevip.WithLeaseVIPs(run.LeaseAnnotations, run.Config.InstanceName, run.Config.RoutingProtocol, run.VIPs)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
leaseClient := run.Mgr.KubernetesClient.CoordinationV1().Leases(run.LeaseID.Namespace())
|
||||
// we use the Lease lock type since edits to Leases are less common
|
||||
// and fewer objects in the cluster watch "all Leases".
|
||||
lock := &resourcelock.LeaseLock{
|
||||
baseLock := &resourcelock.LeaseLock{
|
||||
LeaseMeta: metav1.ObjectMeta{
|
||||
Name: run.LeaseID.Name(),
|
||||
Namespace: run.LeaseID.Namespace(),
|
||||
Annotations: run.LeaseAnnotations,
|
||||
Name: run.LeaseID.Name(),
|
||||
Namespace: run.LeaseID.Namespace(),
|
||||
},
|
||||
Client: run.Mgr.KubernetesClient.CoordinationV1(),
|
||||
LockConfig: resourcelock.ResourceLockConfig{
|
||||
Identity: run.Config.NodeName,
|
||||
},
|
||||
}
|
||||
lock := newAnnotatedLeaseLock(baseLock, leaseClient, run.LeaseID.Name(), annotations)
|
||||
|
||||
// start the leader election code loop
|
||||
leaderelection.RunOrDie(ctx, leaderelection.LeaderElectionConfig{
|
||||
@@ -108,6 +110,7 @@ func runKubernetesLeaderElectionOrDie(ctx context.Context, run *RunConfig) {
|
||||
OnNewLeader: run.OnNewLeader,
|
||||
},
|
||||
})
|
||||
return nil
|
||||
}
|
||||
|
||||
func runEtcdLeaderElectionOrDie(ctx context.Context, run *RunConfig) error {
|
||||
@@ -138,6 +141,7 @@ type RunConfig struct {
|
||||
LeaseID lease.ID
|
||||
Mgr *Manager
|
||||
LeaseAnnotations map[string]string
|
||||
VIPs []string
|
||||
|
||||
// onStartedLeading is called when this member starts leading.
|
||||
OnStartedLeading func(context.Context)
|
||||
@@ -199,7 +203,7 @@ func (em *Manager) NodeWatcher(ctx context.Context, lb *loadbalancer.IPVSLoadBal
|
||||
err = lb.AddBackend(node.Status.Addresses[x].Address, port)
|
||||
if err != nil {
|
||||
log.Error("adding node to load balancer", "node", node.Name, "ip", node.Status.Addresses[x].Address, "err", err)
|
||||
if errors.Is(err, &utils.PanicError{}) {
|
||||
if utils.IsPanicError(err) {
|
||||
return fmt.Errorf("add IPVS backend: %w", err)
|
||||
}
|
||||
}
|
||||
@@ -233,23 +237,20 @@ func (em *Manager) NodeWatcher(ctx context.Context, lb *loadbalancer.IPVSLoadBal
|
||||
// Un-used
|
||||
case watch.Error:
|
||||
log.Error("Error attempting to watch Kubernetes Nodes")
|
||||
|
||||
// This round trip allows us to handle unstructured status
|
||||
errObject := apierrors.FromObject(event.Object)
|
||||
statusErr, ok := errObject.(*apierrors.StatusError)
|
||||
if !ok {
|
||||
log.Error(spew.Sprintf("Received an error which is not *metav1.Status but %#+v", event.Object))
|
||||
}
|
||||
|
||||
status := statusErr.ErrStatus
|
||||
log.Error("watcher", "status", status)
|
||||
watchErr = fmt.Errorf("node watcher error, status: %s", status.String())
|
||||
watchErr = fmt.Errorf("node watcher error: %w", utils.WatchError(event.Object))
|
||||
log.Error("watcher", "err", watchErr)
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
log.Info("Exiting Node watcher")
|
||||
return watchErr
|
||||
if watchErr != nil {
|
||||
return watchErr
|
||||
}
|
||||
if ctx.Err() != nil {
|
||||
return nil
|
||||
}
|
||||
return utils.NewPanicError("node watcher channel closed unexpectedly")
|
||||
}
|
||||
|
||||
func checkIfNodeIsReady(node *v1.Node) bool {
|
||||
|
||||
90
pkg/election/lease_lock.go
Normal file
90
pkg/election/lease_lock.go
Normal file
@@ -0,0 +1,90 @@
|
||||
package election
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
coordinationv1client "k8s.io/client-go/kubernetes/typed/coordination/v1"
|
||||
"k8s.io/client-go/tools/leaderelection/resourcelock"
|
||||
"k8s.io/client-go/util/retry"
|
||||
)
|
||||
|
||||
type annotatedLeaseLock struct {
|
||||
resourcelock.Interface
|
||||
leases coordinationv1client.LeaseInterface
|
||||
name string
|
||||
annotations map[string]string
|
||||
}
|
||||
|
||||
func newAnnotatedLeaseLock(lock resourcelock.Interface, leases coordinationv1client.LeaseInterface,
|
||||
name string, annotations map[string]string) resourcelock.Interface {
|
||||
return &annotatedLeaseLock{Interface: lock, leases: leases, name: name, annotations: annotations}
|
||||
}
|
||||
|
||||
func (lock *annotatedLeaseLock) Get(ctx context.Context) (*resourcelock.LeaderElectionRecord, []byte, error) {
|
||||
return lock.Interface.Get(ctx)
|
||||
}
|
||||
|
||||
func (lock *annotatedLeaseLock) Create(ctx context.Context, record resourcelock.LeaderElectionRecord) error {
|
||||
if err := lock.Interface.Create(ctx, record); err != nil {
|
||||
return err
|
||||
}
|
||||
if record.HolderIdentity != lock.Identity() {
|
||||
return nil
|
||||
}
|
||||
changed, err := lock.ensureAnnotations(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if changed {
|
||||
_, _, err = lock.Interface.Get(ctx)
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
func (lock *annotatedLeaseLock) Update(ctx context.Context, record resourcelock.LeaderElectionRecord) error {
|
||||
if err := lock.Interface.Update(ctx, record); err != nil {
|
||||
return err
|
||||
}
|
||||
if record.HolderIdentity != lock.Identity() {
|
||||
return nil
|
||||
}
|
||||
changed, err := lock.ensureAnnotations(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if changed {
|
||||
_, _, err = lock.Interface.Get(ctx)
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
func (lock *annotatedLeaseLock) ensureAnnotations(ctx context.Context) (bool, error) {
|
||||
changed := false
|
||||
err := retry.RetryOnConflict(retry.DefaultRetry, func() error {
|
||||
resource, err := lock.leases.Get(ctx, lock.name, metav1.GetOptions{})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if resource.Annotations == nil {
|
||||
resource.Annotations = make(map[string]string, len(lock.annotations))
|
||||
}
|
||||
resourceChanged := false
|
||||
for key, value := range lock.annotations {
|
||||
if resource.Annotations[key] == value {
|
||||
continue
|
||||
}
|
||||
resource.Annotations[key] = value
|
||||
resourceChanged = true
|
||||
}
|
||||
if !resourceChanged {
|
||||
return nil
|
||||
}
|
||||
_, err = lock.leases.Update(ctx, resource, metav1.UpdateOptions{})
|
||||
if err == nil {
|
||||
changed = true
|
||||
}
|
||||
return err
|
||||
})
|
||||
return changed, err
|
||||
}
|
||||
139
pkg/election/lease_lock_test.go
Normal file
139
pkg/election/lease_lock_test.go
Normal file
@@ -0,0 +1,139 @@
|
||||
package election
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/client-go/kubernetes/fake"
|
||||
"k8s.io/client-go/tools/leaderelection/resourcelock"
|
||||
)
|
||||
|
||||
func TestAnnotatedLeaseLockPersistsAnnotationsOnCreateAndUpdate(t *testing.T) {
|
||||
client := fake.NewSimpleClientset()
|
||||
leaseClient := client.CoordinationV1().Leases("default")
|
||||
base := &resourcelock.LeaseLock{
|
||||
LeaseMeta: metav1.ObjectMeta{Name: "lease", Namespace: "default"},
|
||||
Client: client.CoordinationV1(),
|
||||
LockConfig: resourcelock.ResourceLockConfig{
|
||||
Identity: "node-a",
|
||||
},
|
||||
}
|
||||
annotations, err := kubevip.WithLeaseVIPs(map[string]string{"example.test/preserved": "true"},
|
||||
"release_a", 248, []string{"192.0.2.10"})
|
||||
if err != nil {
|
||||
t.Fatalf("WithLeaseVIPs() error = %v", err)
|
||||
}
|
||||
lock := newAnnotatedLeaseLock(base, leaseClient, "lease", annotations)
|
||||
record := resourcelock.LeaderElectionRecord{HolderIdentity: "node-a"}
|
||||
if err := lock.Create(context.Background(), record); err != nil {
|
||||
t.Fatalf("Create() error = %v", err)
|
||||
}
|
||||
if err := lock.Update(context.Background(), record); err != nil {
|
||||
t.Fatalf("Update() error = %v", err)
|
||||
}
|
||||
|
||||
resource, err := leaseClient.Get(context.Background(), "lease", metav1.GetOptions{})
|
||||
if err != nil {
|
||||
t.Fatalf("get Lease: %v", err)
|
||||
}
|
||||
value, err := kubevip.ParseLeaseVIPs(resource.Annotations[kubevip.LeaseVIPs])
|
||||
if err != nil {
|
||||
t.Fatalf("ParseLeaseVIPs() error = %v", err)
|
||||
}
|
||||
if value.InstanceName != "release_a" || value.IFAProto != 248 || len(value.VIPs) != 1 ||
|
||||
value.VIPs[0] != (kubevip.LeaseVIP{Index: 0, Value: "192.0.2.10"}) {
|
||||
t.Fatalf("Lease VIP metadata = %+v", value)
|
||||
}
|
||||
if resource.Annotations["example.test/preserved"] != "true" {
|
||||
t.Fatal("Lease update dropped a configured annotation")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotatedLeaseLockFollowerDoesNotOverwriteAnnotations(t *testing.T) {
|
||||
client := fake.NewSimpleClientset()
|
||||
leaseClient := client.CoordinationV1().Leases("default")
|
||||
newBase := func(identity string) *resourcelock.LeaseLock {
|
||||
return &resourcelock.LeaseLock{
|
||||
LeaseMeta: metav1.ObjectMeta{Name: "lease", Namespace: "default"},
|
||||
Client: client.CoordinationV1(),
|
||||
LockConfig: resourcelock.ResourceLockConfig{Identity: identity},
|
||||
}
|
||||
}
|
||||
ownerBase := newBase("node-a")
|
||||
followerBase := newBase("node-b")
|
||||
active, err := kubevip.WithLeaseVIPs(nil, "release_a", 248, []string{"192.0.2.10"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
creator := newAnnotatedLeaseLock(ownerBase, leaseClient, "lease", active)
|
||||
if err := creator.Create(context.Background(), resourcelock.LeaderElectionRecord{HolderIdentity: "node-a"}); err != nil {
|
||||
t.Fatalf("Create() error = %v", err)
|
||||
}
|
||||
|
||||
follower, err := kubevip.WithLeaseVIPs(nil, "release_b", 249, []string{"192.0.2.20"})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
observer := newAnnotatedLeaseLock(followerBase, leaseClient, "lease", follower)
|
||||
if _, _, err := observer.Get(context.Background()); err != nil {
|
||||
t.Fatalf("Get() error = %v", err)
|
||||
}
|
||||
resource, err := leaseClient.Get(context.Background(), "lease", metav1.GetOptions{})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
metadata, err := kubevip.ParseLeaseVIPs(resource.Annotations[kubevip.LeaseVIPs])
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if metadata.InstanceName != "release_a" || metadata.IFAProto != 248 {
|
||||
t.Fatalf("follower overwrote active metadata: %+v", metadata)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnnotatedLeaseLockReleaseDoesNotOverwriteSuccessorMetadata(t *testing.T) {
|
||||
client := fake.NewSimpleClientset()
|
||||
leaseClient := client.CoordinationV1().Leases("default")
|
||||
newLock := func(identity, instanceName string, protocol int, vip string) resourcelock.Interface {
|
||||
base := &resourcelock.LeaseLock{
|
||||
LeaseMeta: metav1.ObjectMeta{Name: "lease", Namespace: "default"},
|
||||
Client: client.CoordinationV1(),
|
||||
LockConfig: resourcelock.ResourceLockConfig{Identity: identity},
|
||||
}
|
||||
annotations, err := kubevip.WithLeaseVIPs(nil, instanceName, protocol, []string{vip})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return newAnnotatedLeaseLock(base, leaseClient, "lease", annotations)
|
||||
}
|
||||
|
||||
first := newLock("node-a", "release_a", 248, "192.0.2.10")
|
||||
if err := first.Create(context.Background(), resourcelock.LeaderElectionRecord{HolderIdentity: "node-a"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := first.Update(context.Background(), resourcelock.LeaderElectionRecord{}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
second := newLock("node-b", "release_b", 249, "192.0.2.20")
|
||||
if _, _, err := second.Get(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := second.Update(context.Background(), resourcelock.LeaderElectionRecord{HolderIdentity: "node-b"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
resource, err := leaseClient.Get(context.Background(), "lease", metav1.GetOptions{})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
metadata, err := kubevip.ParseLeaseVIPs(resource.Annotations[kubevip.LeaseVIPs])
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if metadata.InstanceName != "release_b" || metadata.IFAProto != 249 || metadata.VIPs[0].Value != "192.0.2.20" {
|
||||
t.Fatalf("successor metadata = %+v", metadata)
|
||||
}
|
||||
}
|
||||
152
pkg/endpoints/cleanup.go
Normal file
152
pkg/endpoints/cleanup.go
Normal file
@@ -0,0 +1,152 @@
|
||||
package endpoints
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"sync"
|
||||
|
||||
log "log/slog"
|
||||
|
||||
v1 "k8s.io/api/core/v1"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/bgp"
|
||||
"github.com/kube-vip/kube-vip/pkg/egress"
|
||||
"github.com/kube-vip/kube-vip/pkg/instance"
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
"github.com/kube-vip/kube-vip/pkg/nftables"
|
||||
"github.com/kube-vip/kube-vip/pkg/route"
|
||||
"github.com/kube-vip/kube-vip/pkg/utils"
|
||||
"github.com/kube-vip/kube-vip/pkg/wireguard"
|
||||
)
|
||||
|
||||
// CleanupService stops one Service's datapath before its instance is detached.
|
||||
// The service processor owns labels and instance bookkeeping; this package owns
|
||||
// endpoint-dependent networking and waits for worker shutdown to complete.
|
||||
func CleanupService(ctx context.Context, config *kubevip.Config, bgpServer *bgp.Server, routeMgr *route.Manager,
|
||||
tunnelMgr *wireguard.TunnelManager, serviceInstance *instance.Instance, remaining []*instance.Instance) error {
|
||||
if serviceInstance == nil || serviceInstance.ServiceSnapshot == nil {
|
||||
return nil
|
||||
}
|
||||
service := serviceInstance.ServiceSnapshot
|
||||
for _, serviceCluster := range serviceInstance.Clusters {
|
||||
for _, network := range serviceCluster.Network {
|
||||
network.SetHasEndpoints(false)
|
||||
}
|
||||
}
|
||||
|
||||
if config.EnableBGP {
|
||||
ClearBGPHostsByInstance(ctx, serviceInstance, bgpServer)
|
||||
}
|
||||
if config.EnableRoutingTable {
|
||||
for _, err := range ClearRoutesByInstance(service, serviceInstance, &remaining, routeMgr) {
|
||||
log.Error("unable to clear routes", "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
internalNftablesEgress := service.Annotations[kubevip.EgressInternal] != "" || config.EgressWithNftables
|
||||
if service.Annotations[kubevip.Egress] == "true" && internalNftablesEgress {
|
||||
if err := nftables.DeleteSNATFromAllTables(string(serviceInstance.UID())); err != nil {
|
||||
log.Error("[service] nftables egress teardown", "service", service.Name, "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
sharedVIPs := sharedServiceVIPs(config, serviceInstance, remaining)
|
||||
for _, serviceCluster := range serviceInstance.Clusters {
|
||||
preserve := make([]string, 0, len(serviceCluster.Network))
|
||||
for _, network := range serviceCluster.Network {
|
||||
if _, shared := sharedVIPs[network.IP()]; shared {
|
||||
preserve = append(preserve, network.IP())
|
||||
}
|
||||
}
|
||||
if len(preserve) != 0 {
|
||||
serviceCluster.StopAndWaitPreserving(preserve...)
|
||||
} else {
|
||||
serviceCluster.StopAndWait()
|
||||
}
|
||||
}
|
||||
if err := serviceInstance.CleanupLinkAttachments(remaining...); err != nil {
|
||||
return fmt.Errorf("clean Service link attachments: %w", err)
|
||||
}
|
||||
if service.Annotations[kubevip.Egress] == "true" && !internalNftablesEgress && service.Annotations[kubevip.ActiveEndpoint] != "" {
|
||||
if err := egress.Teardown(service.Annotations[kubevip.ActiveEndpoint], service.Spec.LoadBalancerIP, service.Namespace,
|
||||
string(serviceInstance.UID()), service.Annotations, config.EgressWithNftables); err != nil {
|
||||
log.Error("[service] egress teardown", "err", err)
|
||||
}
|
||||
}
|
||||
if config.EnableWireguard {
|
||||
cleanupWireguardService(tunnelMgr, service)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// StartService starts a Service's cluster datapath after endpoint handling has
|
||||
// made the Service eligible for activation.
|
||||
func StartService(ctx context.Context, service *v1.Service, serviceInstance *instance.Instance, bgpServer *bgp.Server,
|
||||
wg *sync.WaitGroup) error {
|
||||
if serviceInstance == nil {
|
||||
return fmt.Errorf("missing service instance for %s/%s", service.Namespace, service.Name)
|
||||
}
|
||||
for index := range serviceInstance.VIPConfigs {
|
||||
if err := serviceInstance.Clusters[index].StartLoadBalancerService(ctx, serviceInstance.VIPConfigs[index], bgpServer,
|
||||
lease.ServiceNamespacedName(service), wg); err != nil {
|
||||
return fmt.Errorf("start load balancer: %w", err)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func sharedServiceVIPs(config *kubevip.Config, serviceInstance *instance.Instance, remaining []*instance.Instance) map[string]struct{} {
|
||||
shared := make(map[string]struct{})
|
||||
if serviceInstance.ServiceSnapshot == nil ||
|
||||
serviceInstance.ServiceSnapshot.Spec.ExternalTrafficPolicy != v1.ServiceExternalTrafficPolicyTypeCluster {
|
||||
return shared
|
||||
}
|
||||
serviceNamespace, serviceLeaseName := lease.ServiceName(serviceInstance.ServiceSnapshot)
|
||||
serviceLease := lease.NewID(config.LeaderElectionType, serviceNamespace, serviceLeaseName).NamespacedName()
|
||||
addresses := serviceInstance.Addresses()
|
||||
for _, candidate := range remaining {
|
||||
candidateInfo, ok := candidate.CleanupInfo()
|
||||
if !ok || candidateInfo.ExternalTrafficPolicy != v1.ServiceExternalTrafficPolicyTypeCluster {
|
||||
continue
|
||||
}
|
||||
candidateNamespace, candidateLeaseName := lease.ServiceNameFor(candidateInfo.Namespace, candidateInfo.Name, candidateInfo.Lease)
|
||||
candidateLease := lease.NewID(config.LeaderElectionType, candidateNamespace, candidateLeaseName).NamespacedName()
|
||||
if config.EnableServicesElection && candidateLease != serviceLease {
|
||||
continue
|
||||
}
|
||||
for _, address := range candidate.Addresses() {
|
||||
for _, serviceAddress := range addresses {
|
||||
if address == serviceAddress {
|
||||
shared[address] = struct{}{}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return shared
|
||||
}
|
||||
|
||||
func cleanupWireguardService(tunnelMgr *wireguard.TunnelManager, service *v1.Service) {
|
||||
if tunnelMgr == nil {
|
||||
return
|
||||
}
|
||||
forEachServiceDNATChain(service, func(ipv6 bool, serviceID string) {
|
||||
if err := nftables.DeleteIngressChains(ipv6, serviceID); err != nil {
|
||||
log.Error("[wireguard] failed to delete DNAT chains", "ipv6", ipv6, "service", service.Name, "err", err)
|
||||
}
|
||||
})
|
||||
releaseWireguardServiceTunnels(tunnelMgr, service)
|
||||
}
|
||||
|
||||
type wireguardTunnelReleaser interface {
|
||||
ReleaseTunnelForVIP(vip, owner string) error
|
||||
}
|
||||
|
||||
func releaseWireguardServiceTunnels(tunnelMgr wireguardTunnelReleaser, service *v1.Service) {
|
||||
serviceIPs, _ := utils.FetchServiceIPs(service)
|
||||
for _, serviceIP := range serviceIPs {
|
||||
if err := tunnelMgr.ReleaseTunnelForVIP(serviceIP, string(service.UID)); err != nil {
|
||||
log.Error("[wireguard] failed to tear down tunnel", "service", service.Name, "vip", serviceIP, "err", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -6,8 +6,6 @@ import (
|
||||
"net"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
log "log/slog"
|
||||
|
||||
@@ -21,127 +19,171 @@ import (
|
||||
"github.com/kube-vip/kube-vip/pkg/utils"
|
||||
"github.com/kube-vip/kube-vip/pkg/wireguard"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"k8s.io/apimachinery/pkg/watch"
|
||||
"k8s.io/client-go/kubernetes"
|
||||
)
|
||||
|
||||
type Processor struct {
|
||||
config *kubevip.Config
|
||||
provider providers.Provider
|
||||
bgpServer *bgp.Server
|
||||
worker endpointWorker
|
||||
instances *[]*instance.Instance
|
||||
leaseMgr *lease.Manager
|
||||
config *kubevip.Config
|
||||
provider providers.Provider
|
||||
bgpServer *bgp.Server
|
||||
worker endpointWorker
|
||||
instances *[]*instance.Instance
|
||||
instancesMutex *sync.RWMutex
|
||||
leaseMgr *lease.Manager
|
||||
lockService func(types.UID) func()
|
||||
}
|
||||
|
||||
func NewEndpointProcessor(config *kubevip.Config, provider providers.Provider, bgpServer *bgp.Server,
|
||||
instances *[]*instance.Instance, leaseMgr *lease.Manager, tunnelMgr *wireguard.TunnelManager, routeMgr *route.Manager) *Processor {
|
||||
instances *[]*instance.Instance, instancesMutex *sync.RWMutex, leaseMgr *lease.Manager, tunnelMgr *wireguard.TunnelManager, routeMgr *route.Manager,
|
||||
lockService func(types.UID) func()) *Processor {
|
||||
return &Processor{
|
||||
config: config,
|
||||
provider: provider,
|
||||
bgpServer: bgpServer,
|
||||
instances: instances,
|
||||
leaseMgr: leaseMgr,
|
||||
worker: newEndpointWorker(config, provider, bgpServer, instances, leaseMgr, tunnelMgr, routeMgr),
|
||||
config: config,
|
||||
provider: provider,
|
||||
bgpServer: bgpServer,
|
||||
instances: instances,
|
||||
instancesMutex: instancesMutex,
|
||||
leaseMgr: leaseMgr,
|
||||
lockService: lockService,
|
||||
worker: newEndpointWorker(config, provider, bgpServer, leaseMgr, tunnelMgr, routeMgr),
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Processor) AddOrModify(svcCtx *servicecontext.Context, event watch.Event,
|
||||
// Reconcile applies a watch event to the provider and reconciles the service
|
||||
// against the endpoints that remain afterwards. A deleted object is only one of
|
||||
// potentially several backing the service, so deletions are recomputed rather
|
||||
// than assumed to empty it. It reports whether the caller should skip this event
|
||||
// and wait for the next one.
|
||||
func (p *Processor) Reconcile(svcCtx *servicecontext.Context, event watch.Event,
|
||||
lastKnownGoodEndpoint *string, service *v1.Service, id string,
|
||||
serviceFunc func(*servicecontext.Context, *v1.Service, *sync.WaitGroup, bool) error, wg *sync.WaitGroup,
|
||||
wg *sync.WaitGroup,
|
||||
clientSet *kubernetes.Clientset,
|
||||
egressUpdateFunc func(context.Context, *v1.Service) error) (bool, error) {
|
||||
|
||||
var err error
|
||||
if err = p.provider.LoadObject(event.Object, svcCtx.Cancel); err != nil {
|
||||
return false, fmt.Errorf("[%s] error loading k8s object: %w", p.provider.GetLabel(), err)
|
||||
egressUpdateFunc func(context.Context, *v1.Service, *instance.Instance) error) (bool, error) {
|
||||
if p.lockService == nil {
|
||||
return false, fmt.Errorf("service operation lock is not configured")
|
||||
}
|
||||
endpointCount := 0
|
||||
var readinessLossGeneration uint64
|
||||
clearNoEndpoints := false
|
||||
updatedService, inst, changed, skip, err := func() (*v1.Service, *instance.Instance, bool, bool, error) {
|
||||
unlockService := p.lockService(service.UID)
|
||||
defer unlockService()
|
||||
|
||||
endpoints, err := p.worker.getEndpoints(service, id)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
if err := p.applyEvent(svcCtx, event); err != nil {
|
||||
return nil, nil, false, false, err
|
||||
}
|
||||
|
||||
if err := p.worker.setInstanceEndpointsStatus(svcCtx.Ctx, service, endpoints); err != nil {
|
||||
log.Error("updating instance", "err", err)
|
||||
}
|
||||
|
||||
allowReconcileWithoutEndpoints := shouldAllowReconcileWithoutEndpoints(service)
|
||||
|
||||
// Find out if we have any local endpoints
|
||||
// if out endpoint is empty then populate it
|
||||
// if not, go through the endpoints and see if ours still exists
|
||||
// If we have a local endpoint then begin the leader Election, unless it's already running
|
||||
//
|
||||
|
||||
les := atomic.Int64{}
|
||||
|
||||
// Check that we have local endpoints
|
||||
if len(endpoints) != 0 {
|
||||
// Ignore IPv4
|
||||
endpoints, err := p.worker.getEndpoints(service, id)
|
||||
if err != nil {
|
||||
return nil, nil, false, false, fmt.Errorf("[%s] error getting endpoints: %w", p.provider.GetLabel(), err)
|
||||
}
|
||||
if service.Annotations[kubevip.EgressIPv6] == "true" && !hasV6(endpoints) {
|
||||
return true, nil
|
||||
endpoints = nil
|
||||
}
|
||||
endpointCount = len(endpoints)
|
||||
|
||||
inst := p.findServiceInstance(service)
|
||||
|
||||
if err := p.worker.setInstanceEndpointsStatus(service, inst, endpoints); err != nil {
|
||||
log.Error("updating instance", "err", err)
|
||||
}
|
||||
|
||||
p.updateLastKnownGoodEndpoint(lastKnownGoodEndpoint, endpoints, service)
|
||||
allowReconcileWithoutEndpoints := shouldAllowReconcileWithoutEndpoints(service)
|
||||
|
||||
if err := p.startServiceHandlingIfNeeded(svcCtx, service, serviceFunc, wg, &les); err != nil {
|
||||
return true, err
|
||||
}
|
||||
if len(endpoints) != 0 {
|
||||
p.updateLastKnownGoodEndpoint(lastKnownGoodEndpoint, endpoints, service)
|
||||
|
||||
svcCtx.SignalReadiness()
|
||||
|
||||
// There are local endpoints available on the node
|
||||
// Process immediately if:
|
||||
// - No services/leader election is enabled, OR
|
||||
// - WireGuard is enabled (it always needs immediate DNAT rule updates)
|
||||
if (!p.config.EnableServicesElection && !p.config.EnableLeaderElection) || p.config.EnableWireguard {
|
||||
if err := p.worker.processInstance(svcCtx, service); err != nil {
|
||||
return false, fmt.Errorf("failed to process non-empty instance: %w", err)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if allowReconcileWithoutEndpoints {
|
||||
// Explicit opt-in for controllers that create LoadBalancer services without endpoints
|
||||
if err := p.startServiceHandlingIfNeeded(svcCtx, service, serviceFunc, wg, &les); err != nil {
|
||||
return true, err
|
||||
if err := p.startServiceHandlingIfNeeded(svcCtx, service, inst, wg); err != nil {
|
||||
return nil, nil, false, true, err
|
||||
}
|
||||
|
||||
svcCtx.SignalReadiness()
|
||||
|
||||
if (!p.config.EnableServicesElection && !p.config.EnableLeaderElection) || p.config.EnableWireguard {
|
||||
if err := p.worker.processInstance(svcCtx, service); err != nil {
|
||||
return false, fmt.Errorf("failed to process endpointless instance: %w", err)
|
||||
if p.shouldProcessInstance() {
|
||||
if err := p.worker.processInstance(svcCtx.Ctx, &svcCtx.ConfiguredNetworks, service, inst); err != nil {
|
||||
return nil, nil, false, false, fmt.Errorf("failed to process non-empty instance: %w", err)
|
||||
}
|
||||
}
|
||||
} else if svcCtx.Signalled.Load() {
|
||||
// There are no local endpoints
|
||||
svcCtx.ResetReadiness()
|
||||
p.worker.clear(svcCtx, lastKnownGoodEndpoint, service)
|
||||
if p.config.EnableARP && !p.config.EnableServicesElection {
|
||||
i := instance.FindServiceInstance(service, *p.instances)
|
||||
for _, c := range i.Clusters {
|
||||
c.Stop()
|
||||
} else {
|
||||
if allowReconcileWithoutEndpoints {
|
||||
// Explicit opt-in for controllers that create LoadBalancer services without endpoints
|
||||
if err := p.startServiceHandlingIfNeeded(svcCtx, service, inst, wg); err != nil {
|
||||
return nil, nil, false, true, err
|
||||
}
|
||||
svcCtx.SignalReadiness()
|
||||
|
||||
if p.shouldProcessInstance() {
|
||||
if err := p.worker.processInstance(svcCtx.Ctx, &svcCtx.ConfiguredNetworks, service, inst); err != nil {
|
||||
return nil, nil, false, false, fmt.Errorf("failed to process endpointless instance: %w", err)
|
||||
}
|
||||
}
|
||||
} else if svcCtx.IsReady() {
|
||||
readinessLossGeneration, _, _, _ = svcCtx.ReadinessState()
|
||||
clearNoEndpoints = true
|
||||
}
|
||||
}
|
||||
|
||||
updatedService, changed := p.updateAnnotations(service, inst, lastKnownGoodEndpoint, clientSet)
|
||||
return updatedService, inst, changed, false, nil
|
||||
}()
|
||||
if err != nil || skip {
|
||||
return skip, err
|
||||
}
|
||||
if clearNoEndpoints && svcCtx.ResetReadinessGeneration(readinessLossGeneration) {
|
||||
unlockService := p.lockService(service.UID)
|
||||
p.handleNoEndpoints(svcCtx, service, inst, lastKnownGoodEndpoint)
|
||||
unlockService()
|
||||
}
|
||||
|
||||
if changed && egressUpdateFunc != nil {
|
||||
if err := egressUpdateFunc(context.Background(), updatedService, inst); err != nil {
|
||||
log.Error("failed to reconfigure egress", "service", service.Name, "namespace", service.Namespace, "err", err)
|
||||
}
|
||||
}
|
||||
|
||||
// Set the service accordingly
|
||||
p.updateAnnotations(service, lastKnownGoodEndpoint, clientSet, egressUpdateFunc)
|
||||
|
||||
log.Debug("watcher", "provider",
|
||||
p.provider.GetLabel(), "service name", service.Name, "namespace", service.Namespace, "endpoints", len(endpoints), "last endpoint", *lastKnownGoodEndpoint)
|
||||
p.provider.GetLabel(), "service name", service.Name, "namespace", service.Namespace, "endpoints", endpointCount, "last endpoint", *lastKnownGoodEndpoint)
|
||||
|
||||
return false, nil
|
||||
}
|
||||
|
||||
func (p *Processor) Delete(ctx context.Context, service *v1.Service, id string) error {
|
||||
if err := p.worker.delete(ctx, service, id); err != nil {
|
||||
return fmt.Errorf("[%s] error deleting service: %w", p.provider.GetLabel(), err)
|
||||
// applyEvent updates the provider's view of the objects backing this service.
|
||||
func (p *Processor) applyEvent(svcCtx *servicecontext.Context, event watch.Event) error {
|
||||
if event.Type == watch.Deleted {
|
||||
if err := p.provider.DeleteObject(event.Object); err != nil {
|
||||
return fmt.Errorf("[%s] error deleting k8s object: %w", p.provider.GetLabel(), err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
if err := p.provider.LoadObject(event.Object, svcCtx.Cancel); err != nil {
|
||||
return fmt.Errorf("[%s] error loading k8s object: %w", p.provider.GetLabel(), err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// shouldProcessInstance reports whether this node has to program the datapath
|
||||
// itself, rather than waiting to be told to by a leader election callback.
|
||||
// WireGuard always reprograms, because its DNAT rules are per-endpoint.
|
||||
func (p *Processor) shouldProcessInstance() bool {
|
||||
return (!p.config.EnableServicesElection && !p.config.EnableLeaderElection) || p.config.EnableWireguard
|
||||
}
|
||||
|
||||
// handleNoEndpoints tears down everything backing a service that no longer has
|
||||
// any usable endpoints.
|
||||
func (p *Processor) handleNoEndpoints(svcCtx *servicecontext.Context, service *v1.Service, inst *instance.Instance, lastKnownGoodEndpoint *string) {
|
||||
p.worker.clear(svcCtx.Ctx, &svcCtx.ConfiguredNetworks, lastKnownGoodEndpoint, service, inst)
|
||||
stopWorkers := p.config.EnableARP || (p.config.EnableRoutingTable && p.config.EnableLeaderElection)
|
||||
if stopWorkers && !p.config.EnableServicesElection {
|
||||
if inst != nil {
|
||||
for _, c := range inst.Clusters {
|
||||
c.StopAndWait()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Processor) updateLastKnownGoodEndpoint(lastKnownGoodEndpoint *string, endpoints []string, service *v1.Service) {
|
||||
// if we haven't populated one, then do so
|
||||
family := utils.IPv4Family
|
||||
@@ -176,33 +218,44 @@ func (p *Processor) updateLastKnownGoodEndpoint(lastKnownGoodEndpoint *string, e
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Processor) updateAnnotations(service *v1.Service, lastKnownGoodEndpoint *string,
|
||||
clientSet *kubernetes.Clientset,
|
||||
egressUpdateFunc func(context.Context, *v1.Service) error) {
|
||||
func (p *Processor) updateAnnotations(service *v1.Service, inst *instance.Instance, lastKnownGoodEndpoint *string,
|
||||
clientSet *kubernetes.Clientset) (*v1.Service, bool) {
|
||||
// Set the service accordingly
|
||||
if service.Annotations[kubevip.Egress] == "true" {
|
||||
ip := net.ParseIP(*lastKnownGoodEndpoint)
|
||||
if *lastKnownGoodEndpoint != "" {
|
||||
ip := net.ParseIP(*lastKnownGoodEndpoint)
|
||||
expectIPv6 := service.Annotations[kubevip.EgressIPv6] == "true"
|
||||
if ip == nil || (ip.To4() == nil) != expectIPv6 {
|
||||
log.Warn("ignoring active endpoint with unexpected address family",
|
||||
"service", service.Name,
|
||||
"namespace", service.Namespace,
|
||||
"endpoint", *lastKnownGoodEndpoint,
|
||||
"expected_ipv6", expectIPv6)
|
||||
return nil, false
|
||||
}
|
||||
}
|
||||
|
||||
// Store old values from ServiceSnapshot to detect if annotation actually changed
|
||||
// We use the ServiceSnapshot instead of the service parameter because the service parameter
|
||||
// may have stale annotations if the last update failed
|
||||
var oldEndpoint, oldEndpointIPv6 string
|
||||
if p.instances != nil {
|
||||
serviceInstance := instance.FindServiceInstance(service, *p.instances)
|
||||
if serviceInstance != nil {
|
||||
oldEndpoint = serviceInstance.ServiceSnapshot.Annotations[kubevip.ActiveEndpoint]
|
||||
oldEndpointIPv6 = serviceInstance.ServiceSnapshot.Annotations[kubevip.ActiveEndpointIPv6]
|
||||
snapshotFound := false
|
||||
if inst != nil {
|
||||
if inst.ServiceSnapshot != nil {
|
||||
snapshotFound = true
|
||||
oldEndpoint = inst.ServiceSnapshot.Annotations[kubevip.ActiveEndpoint]
|
||||
oldEndpointIPv6 = inst.ServiceSnapshot.Annotations[kubevip.ActiveEndpointIPv6]
|
||||
}
|
||||
}
|
||||
// Fall back to service annotations if we couldn't find the instance
|
||||
if oldEndpoint == "" && oldEndpointIPv6 == "" {
|
||||
// Empty annotations in an existing snapshot are meaningful after a zero-endpoint transition.
|
||||
if !snapshotFound {
|
||||
oldEndpoint = service.Annotations[kubevip.ActiveEndpoint]
|
||||
oldEndpointIPv6 = service.Annotations[kubevip.ActiveEndpointIPv6]
|
||||
}
|
||||
|
||||
// Determine which annotation to update based on IP version
|
||||
var endpoint, endpointIPv6 string
|
||||
if ip.To4() == nil && !p.config.EnableEndpoints {
|
||||
if service.Annotations[kubevip.EgressIPv6] == "true" && !p.config.EnableEndpoints {
|
||||
// IPv6
|
||||
endpointIPv6 = *lastKnownGoodEndpoint
|
||||
endpoint = oldEndpoint // Preserve existing IPv4 if any
|
||||
@@ -215,7 +268,7 @@ func (p *Processor) updateAnnotations(service *v1.Service, lastKnownGoodEndpoint
|
||||
// Check if annotation actually changed
|
||||
annotationChanged := (oldEndpoint != endpoint) || (oldEndpointIPv6 != endpointIPv6)
|
||||
if !annotationChanged {
|
||||
return // Nothing to do
|
||||
return nil, false
|
||||
}
|
||||
|
||||
// Persist to Kubernetes
|
||||
@@ -223,48 +276,32 @@ func (p *Processor) updateAnnotations(service *v1.Service, lastKnownGoodEndpoint
|
||||
|
||||
if err := p.provider.UpdateServiceAnnotation(ctx, endpoint, endpointIPv6, service, clientSet); err != nil {
|
||||
log.Warn("failed to update service annotation", "service", service.Name, "namespace", service.Namespace, "err", err)
|
||||
return
|
||||
return nil, false
|
||||
}
|
||||
|
||||
log.Debug("updated active endpoint annotation", "service", service.Name, "namespace", service.Namespace, "endpoint", *lastKnownGoodEndpoint)
|
||||
|
||||
// Trigger egress reconfiguration
|
||||
// For services with leader election, the service watcher doesn't process Modified events
|
||||
// after initial setup, so we need to directly call the update function
|
||||
if egressUpdateFunc != nil {
|
||||
// Create a copy of service with updated annotations
|
||||
svcCopy := service.DeepCopy()
|
||||
svcCopy.Annotations[kubevip.ActiveEndpoint] = endpoint
|
||||
svcCopy.Annotations[kubevip.ActiveEndpointIPv6] = endpointIPv6
|
||||
|
||||
if err := egressUpdateFunc(ctx, svcCopy); err != nil {
|
||||
log.Error("failed to reconfigure egress", "service", service.Name, "namespace", service.Namespace, "err", err)
|
||||
}
|
||||
}
|
||||
svcCopy := service.DeepCopy()
|
||||
svcCopy.Annotations[kubevip.ActiveEndpoint] = endpoint
|
||||
svcCopy.Annotations[kubevip.ActiveEndpointIPv6] = endpointIPv6
|
||||
return svcCopy, true
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
|
||||
func (p *Processor) startServiceHandlingIfNeeded(svcCtx *servicecontext.Context, service *v1.Service,
|
||||
serviceFunc func(*servicecontext.Context, *v1.Service, *sync.WaitGroup, bool) error, wg *sync.WaitGroup, les *atomic.Int64) error {
|
||||
inst *instance.Instance, wg *sync.WaitGroup) error {
|
||||
if p.config.EnableServicesElection {
|
||||
wg.Go(func() {
|
||||
les.Add(1)
|
||||
p.startLeaderElection(svcCtx, service, serviceFunc, wg)
|
||||
})
|
||||
return nil
|
||||
}
|
||||
|
||||
if p.config.EnableARP || (p.config.EnableRoutingTable && p.config.EnableLeaderElection) {
|
||||
if !svcCtx.Signalled.Load() {
|
||||
inst := instance.FindServiceInstance(service, *p.instances)
|
||||
if !svcCtx.IsReady() {
|
||||
if inst == nil {
|
||||
return fmt.Errorf("[%s] failed to find an instance for service %s/%s", p.provider.GetLabel(), service.Namespace, service.Name)
|
||||
}
|
||||
for x := range inst.VIPConfigs {
|
||||
log.Debug("starting loadbalancer for service", "provider", p.provider.GetLabel(), "name", service.Name, "namespace", service.Namespace, "uid", service.UID)
|
||||
if err := inst.Clusters[x].StartLoadBalancerService(svcCtx.Ctx, inst.VIPConfigs[x], p.bgpServer, lease.ServiceNamespacedName(service), wg); err != nil {
|
||||
return fmt.Errorf("failed to start lb: %w", err)
|
||||
}
|
||||
if err := StartService(svcCtx.Ctx, service, inst, p.bgpServer, wg); err != nil {
|
||||
return fmt.Errorf("start service datapath: %w", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -272,30 +309,16 @@ func (p *Processor) startServiceHandlingIfNeeded(svcCtx *servicecontext.Context,
|
||||
return nil
|
||||
}
|
||||
|
||||
func (p *Processor) startLeaderElection(svcCtx *servicecontext.Context, service *v1.Service, serviceFunc func(*servicecontext.Context, *v1.Service, *sync.WaitGroup, bool) error, wg *sync.WaitGroup) {
|
||||
// This is a blocking function, that will restart (in the event of failure)
|
||||
for {
|
||||
select {
|
||||
case <-svcCtx.Ctx.Done():
|
||||
return
|
||||
default:
|
||||
leaseNamespace, serviceLease := lease.ServiceName(service)
|
||||
id := lease.NewID(p.config.LeaderElectionType, leaseNamespace, serviceLease)
|
||||
l := p.leaseMgr.Get(id)
|
||||
l.Lock()
|
||||
|
||||
if !l.Elected.Load() {
|
||||
l.Unlock()
|
||||
err := serviceFunc(svcCtx, service, wg, true)
|
||||
if err != nil {
|
||||
log.Error(err.Error())
|
||||
}
|
||||
} else {
|
||||
l.Unlock()
|
||||
time.Sleep(time.Millisecond * 200)
|
||||
}
|
||||
}
|
||||
func (p *Processor) findServiceInstance(service *v1.Service) *instance.Instance {
|
||||
if p.instances == nil {
|
||||
return nil
|
||||
}
|
||||
if p.instancesMutex != nil {
|
||||
p.instancesMutex.RLock()
|
||||
defer p.instancesMutex.RUnlock()
|
||||
}
|
||||
inst := instance.FindServiceInstance(service, *p.instances)
|
||||
return inst
|
||||
}
|
||||
|
||||
func shouldAllowReconcileWithoutEndpoints(service *v1.Service) bool {
|
||||
|
||||
@@ -2,13 +2,12 @@ package endpoints
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
log "log/slog"
|
||||
"sync"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/bgp"
|
||||
"github.com/kube-vip/kube-vip/pkg/instance"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
"github.com/kube-vip/kube-vip/pkg/servicecontext"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
)
|
||||
|
||||
@@ -24,19 +23,19 @@ func newBGP(generic generic, bgpServer *bgp.Server) endpointWorker {
|
||||
}
|
||||
}
|
||||
|
||||
func (b *BGP) processInstance(svcCtx *servicecontext.Context, service *v1.Service) error {
|
||||
if instance := instance.FindServiceInstance(service, *b.instances); instance != nil {
|
||||
for _, cluster := range instance.Clusters {
|
||||
func (b *BGP) processInstance(ctx context.Context, configuredNetworks *sync.Map, service *v1.Service, inst *instance.Instance) error {
|
||||
if inst != nil {
|
||||
for _, cluster := range inst.Clusters {
|
||||
for i := range cluster.Network {
|
||||
if !svcCtx.IsNetworkConfigured(cluster.Network[i].IP()) {
|
||||
if _, configured := configuredNetworks.Load(cluster.Network[i].IP()); !configured {
|
||||
log.Debug("attempting to advertise BGP service", "provider", b.provider.GetLabel(), "ip", cluster.Network[i].IP())
|
||||
err := b.bgpServer.AddHost(svcCtx.Ctx, cluster.Network[i].CIDR(), lease.ServiceNamespacedName(service))
|
||||
err := b.bgpServer.AddHost(ctx, cluster.Network[i].CIDR(), lease.ServiceNamespacedName(service))
|
||||
if err != nil {
|
||||
log.Error("error adding BGP host", "provider", b.provider.GetLabel(), "err", err)
|
||||
} else {
|
||||
log.Info("added BGP host", "provider",
|
||||
b.provider.GetLabel(), "ip", cluster.Network[i].CIDR(), "service name", service.Name, "namespace", service.Namespace)
|
||||
svcCtx.ConfiguredNetworks.Store(cluster.Network[i].IP(), true)
|
||||
configuredNetworks.Store(cluster.Network[i].IP(), true)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -45,19 +44,19 @@ func (b *BGP) processInstance(svcCtx *servicecontext.Context, service *v1.Servic
|
||||
return nil
|
||||
}
|
||||
|
||||
func (b *BGP) clear(svcCtx *servicecontext.Context, lastKnownGoodEndpoint *string, service *v1.Service) {
|
||||
func (b *BGP) clear(ctx context.Context, configuredNetworks *sync.Map, lastKnownGoodEndpoint *string, service *v1.Service, inst *instance.Instance) {
|
||||
if !b.config.EnableServicesElection && !b.config.EnableLeaderElection {
|
||||
// If BGP mode is enabled - routes should be deleted
|
||||
if instance := instance.FindServiceInstance(service, *b.instances); instance != nil {
|
||||
for _, cluster := range instance.Clusters {
|
||||
if inst != nil {
|
||||
for _, cluster := range inst.Clusters {
|
||||
for i := range cluster.Network {
|
||||
err := b.bgpServer.DelHost(svcCtx.Ctx, cluster.Network[i].CIDR(), lease.ServiceNamespacedName(service))
|
||||
err := b.bgpServer.DelHost(ctx, cluster.Network[i].CIDR(), lease.ServiceNamespacedName(service))
|
||||
if err != nil {
|
||||
log.Error("deleting BGP host", "provider", b.provider.GetLabel(), "ip", cluster.Network[i].IP(), "err", err)
|
||||
} else {
|
||||
log.Info("deleted BGP host", "provider",
|
||||
b.provider.GetLabel(), "ip", cluster.Network[i].IP(), "service name", service.Name, "namespace", service.Namespace)
|
||||
svcCtx.ConfiguredNetworks.Delete(cluster.Network[i].IP())
|
||||
configuredNetworks.Delete(cluster.Network[i].IP())
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -66,43 +65,13 @@ func (b *BGP) clear(svcCtx *servicecontext.Context, lastKnownGoodEndpoint *strin
|
||||
}
|
||||
|
||||
b.clearEgress(lastKnownGoodEndpoint, service)
|
||||
|
||||
if svcCtx.LeaderCancel != nil {
|
||||
svcCtx.LeaderCancel()
|
||||
}
|
||||
}
|
||||
|
||||
func (b *BGP) getEndpoints(service *v1.Service, id string) ([]string, error) {
|
||||
return b.getAllEndpoints(service, id)
|
||||
}
|
||||
|
||||
func (b *BGP) delete(ctx context.Context, service *v1.Service, id string) error {
|
||||
// When no-leader-elecition mode
|
||||
if !b.config.EnableServicesElection && !b.config.EnableLeaderElection {
|
||||
// find all existing local endpoints
|
||||
endpoints, err := b.getEndpoints(service, id)
|
||||
if err != nil {
|
||||
return fmt.Errorf("[%s] error getting endpoints: %w", b.provider.GetLabel(), err)
|
||||
}
|
||||
|
||||
// If there were local endpoints deleted
|
||||
if len(endpoints) > 0 {
|
||||
b.deleteAction(ctx, service)
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (b *BGP) deleteAction(ctx context.Context, service *v1.Service) {
|
||||
b.clearBGPHosts(ctx, service)
|
||||
}
|
||||
|
||||
func (b *BGP) clearBGPHosts(ctx context.Context, service *v1.Service) {
|
||||
ClearBGPHosts(ctx, service, b.instances, b.bgpServer)
|
||||
}
|
||||
|
||||
func (b *BGP) setInstanceEndpointsStatus(_ context.Context, _ *v1.Service, _ []string) error {
|
||||
func (b *BGP) setInstanceEndpointsStatus(_ *v1.Service, _ *instance.Instance, _ []string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
log "log/slog"
|
||||
"sync"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/bgp"
|
||||
"github.com/kube-vip/kube-vip/pkg/egress"
|
||||
@@ -12,26 +13,24 @@ import (
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
"github.com/kube-vip/kube-vip/pkg/route"
|
||||
"github.com/kube-vip/kube-vip/pkg/servicecontext"
|
||||
"github.com/kube-vip/kube-vip/pkg/wireguard"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
)
|
||||
|
||||
type endpointWorker interface {
|
||||
processInstance(svcCtx *servicecontext.Context, service *v1.Service) error
|
||||
clear(svcCtx *servicecontext.Context, lastKnownGoodEndpoint *string, service *v1.Service)
|
||||
processInstance(ctx context.Context, configuredNetworks *sync.Map, service *v1.Service, inst *instance.Instance) error
|
||||
clear(ctx context.Context, configuredNetworks *sync.Map, lastKnownGoodEndpoint *string, service *v1.Service, inst *instance.Instance)
|
||||
getEndpoints(service *v1.Service, id string) ([]string, error)
|
||||
removeEgress(service *v1.Service, lastKnownGoodEndpoint *string)
|
||||
delete(ctx context.Context, service *v1.Service, id string) error
|
||||
setInstanceEndpointsStatus(ctx context.Context, service *v1.Service, endpoints []string) error
|
||||
setInstanceEndpointsStatus(service *v1.Service, inst *instance.Instance, endpoints []string) error
|
||||
}
|
||||
|
||||
func newEndpointWorker(config *kubevip.Config, provider providers.Provider, bgpServer *bgp.Server, instances *[]*instance.Instance,
|
||||
func newEndpointWorker(config *kubevip.Config, provider providers.Provider, bgpServer *bgp.Server,
|
||||
leaseMgr *lease.Manager, tunnelMgr *wireguard.TunnelManager, routeMgr *route.Manager) endpointWorker {
|
||||
generic := newGeneric(config, provider, instances, leaseMgr)
|
||||
generic := newGeneric(config, provider, leaseMgr)
|
||||
|
||||
if config.EnableWireguard {
|
||||
return newWireguardWorker(config, provider, bgpServer, instances, leaseMgr, tunnelMgr)
|
||||
return newWireguardWorker(config, provider, tunnelMgr)
|
||||
}
|
||||
if config.EnableRoutingTable {
|
||||
return newRoutingTable(generic, routeMgr)
|
||||
@@ -44,30 +43,25 @@ func newEndpointWorker(config *kubevip.Config, provider providers.Provider, bgpS
|
||||
}
|
||||
|
||||
type generic struct {
|
||||
config *kubevip.Config
|
||||
provider providers.Provider
|
||||
instances *[]*instance.Instance
|
||||
leaseMgr *lease.Manager
|
||||
config *kubevip.Config
|
||||
provider providers.Provider
|
||||
leaseMgr *lease.Manager
|
||||
}
|
||||
|
||||
func newGeneric(config *kubevip.Config, provider providers.Provider, instances *[]*instance.Instance, leaseMgr *lease.Manager) generic {
|
||||
func newGeneric(config *kubevip.Config, provider providers.Provider, leaseMgr *lease.Manager) generic {
|
||||
return generic{
|
||||
config: config,
|
||||
provider: provider,
|
||||
instances: instances,
|
||||
leaseMgr: leaseMgr,
|
||||
config: config,
|
||||
provider: provider,
|
||||
leaseMgr: leaseMgr,
|
||||
}
|
||||
}
|
||||
|
||||
func (g *generic) processInstance(_ *servicecontext.Context, _ *v1.Service) error {
|
||||
func (g *generic) processInstance(_ context.Context, _ *sync.Map, _ *v1.Service, _ *instance.Instance) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (g *generic) clear(svcCtx *servicecontext.Context, lastKnownGoodEndpoint *string, service *v1.Service) {
|
||||
func (g *generic) clear(_ context.Context, _ *sync.Map, lastKnownGoodEndpoint *string, service *v1.Service, _ *instance.Instance) {
|
||||
g.clearEgress(lastKnownGoodEndpoint, service)
|
||||
if svcCtx.LeaderCancel != nil {
|
||||
svcCtx.LeaderCancel()
|
||||
}
|
||||
}
|
||||
|
||||
func (g *generic) clearEgress(lastKnownGoodEndpoint *string, service *v1.Service) {
|
||||
@@ -105,10 +99,6 @@ func (g *generic) getAllEndpoints(service *v1.Service, id string) ([]string, err
|
||||
func (g *generic) removeEgress(_ *v1.Service, _ *string) {
|
||||
}
|
||||
|
||||
func (g *generic) delete(_ context.Context, _ *v1.Service, _ string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (g *generic) setInstanceEndpointsStatus(_ context.Context, _ *v1.Service, _ []string) error {
|
||||
func (g *generic) setInstanceEndpointsStatus(_ *v1.Service, _ *instance.Instance, _ []string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -12,13 +12,11 @@ import (
|
||||
"github.com/kube-vip/kube-vip/pkg/instance"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
"github.com/kube-vip/kube-vip/pkg/route"
|
||||
"github.com/kube-vip/kube-vip/pkg/servicecontext"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
)
|
||||
|
||||
type RoutingTable struct {
|
||||
generic
|
||||
mtx sync.Mutex
|
||||
routeMgr *route.Manager
|
||||
}
|
||||
|
||||
@@ -29,19 +27,18 @@ func newRoutingTable(generic generic, routeMgr *route.Manager) endpointWorker {
|
||||
}
|
||||
}
|
||||
|
||||
func (rt *RoutingTable) processInstance(svcCtx *servicecontext.Context, service *v1.Service) error {
|
||||
inst := instance.FindServiceInstance(service, *rt.instances)
|
||||
func (rt *RoutingTable) processInstance(_ context.Context, configuredNetworks *sync.Map, service *v1.Service, inst *instance.Instance) error {
|
||||
if inst != nil {
|
||||
for _, cluster := range inst.Clusters {
|
||||
for i := range cluster.Network {
|
||||
if !svcCtx.IsNetworkConfigured(cluster.Network[i].IP()) && cluster.Network[i].HasEndpoints() {
|
||||
if _, configured := configuredNetworks.Load(cluster.Network[i].IP()); !configured && cluster.Network[i].HasEndpoints() {
|
||||
if err := rt.routeMgr.Add(lease.ServiceNamespacedName(service), cluster.Network[i], false, true); err != nil {
|
||||
return fmt.Errorf("[%s] error adding route: %s", rt.provider.GetLabel(), err.Error())
|
||||
} else {
|
||||
log.Info("added route", "provider",
|
||||
rt.provider.GetLabel(), "ip", cluster.Network[i].IP(), "service name", service.Name, "namespace",
|
||||
service.Namespace, "interface", cluster.Network[i].Interface(), "tableID", rt.config.RoutingTableID)
|
||||
svcCtx.ConfiguredNetworks.Store(cluster.Network[i].IP(), true)
|
||||
configuredNetworks.Store(cluster.Network[i].IP(), true)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -51,12 +48,10 @@ func (rt *RoutingTable) processInstance(svcCtx *servicecontext.Context, service
|
||||
return nil
|
||||
}
|
||||
|
||||
func (rt *RoutingTable) clear(svcCtx *servicecontext.Context, lastKnownGoodEndpoint *string, service *v1.Service) {
|
||||
rt.mtx.Lock()
|
||||
defer rt.mtx.Unlock()
|
||||
func (rt *RoutingTable) clear(_ context.Context, configuredNetworks *sync.Map, lastKnownGoodEndpoint *string, service *v1.Service, inst *instance.Instance) {
|
||||
if !rt.config.EnableServicesElection {
|
||||
if errs := ClearRoutes(service, rt.instances, rt.routeMgr); len(errs) == 0 {
|
||||
svcCtx.ConfiguredNetworks.Clear()
|
||||
if errs := ClearRoutesByInstance(service, inst, nil, rt.routeMgr); len(errs) == 0 {
|
||||
configuredNetworks.Clear()
|
||||
} else {
|
||||
for _, err := range errs {
|
||||
log.Error("error while clearing routes", "err", err)
|
||||
@@ -65,10 +60,6 @@ func (rt *RoutingTable) clear(svcCtx *servicecontext.Context, lastKnownGoodEndpo
|
||||
}
|
||||
|
||||
rt.clearEgress(lastKnownGoodEndpoint, service)
|
||||
|
||||
if svcCtx.LeaderCancel != nil {
|
||||
svcCtx.LeaderCancel()
|
||||
}
|
||||
}
|
||||
|
||||
func (rt *RoutingTable) getEndpoints(service *v1.Service, id string) ([]string, error) {
|
||||
@@ -82,30 +73,7 @@ func (rt *RoutingTable) removeEgress(service *v1.Service, lastKnownGoodEndpoint
|
||||
}
|
||||
}
|
||||
|
||||
func (rt *RoutingTable) delete(_ context.Context, service *v1.Service, id string) error {
|
||||
// When no-leader-elecition mode
|
||||
if !rt.config.EnableServicesElection && !rt.config.EnableLeaderElection {
|
||||
// find all existing local endpoints
|
||||
endpoints, err := rt.getEndpoints(service, id)
|
||||
if err != nil {
|
||||
return fmt.Errorf("[%s] error getting endpoints: %w", rt.provider.GetLabel(), err)
|
||||
}
|
||||
|
||||
// If there were local endpoints deleted
|
||||
if len(endpoints) > 0 {
|
||||
rt.deleteAction(service)
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (rt *RoutingTable) deleteAction(service *v1.Service) {
|
||||
ClearRoutes(service, rt.instances, rt.routeMgr)
|
||||
}
|
||||
|
||||
func (rt *RoutingTable) setInstanceEndpointsStatus(ctx context.Context, service *v1.Service, endpoints []string) error {
|
||||
inst := instance.FindServiceInstance(service, *rt.instances)
|
||||
func (rt *RoutingTable) setInstanceEndpointsStatus(service *v1.Service, inst *instance.Instance, endpoints []string) error {
|
||||
if inst == nil {
|
||||
log.Error("failed to find the instance", "namespace", service.Namespace, "name", service.Name, "uid", service.UID, "provider", rt.provider.GetLabel())
|
||||
} else {
|
||||
|
||||
@@ -5,13 +5,19 @@ import (
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/cluster"
|
||||
"github.com/kube-vip/kube-vip/pkg/endpoints/providers"
|
||||
"github.com/kube-vip/kube-vip/pkg/instance"
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
"github.com/kube-vip/kube-vip/pkg/route"
|
||||
"github.com/kube-vip/kube-vip/pkg/servicecontext"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
discoveryv1 "k8s.io/api/discovery/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
"k8s.io/apimachinery/pkg/watch"
|
||||
"k8s.io/client-go/kubernetes"
|
||||
)
|
||||
|
||||
func TestShouldAllowReconcileWithoutEndpoints(t *testing.T) {
|
||||
@@ -40,27 +46,308 @@ type fakeWorker struct {
|
||||
endpoints []string
|
||||
clearCalled bool
|
||||
processCalled bool
|
||||
processHook func()
|
||||
}
|
||||
|
||||
func (f *fakeWorker) processInstance(_ *servicecontext.Context, _ *v1.Service) error {
|
||||
type annotationUpdate struct {
|
||||
endpoint string
|
||||
endpointIPv6 string
|
||||
}
|
||||
|
||||
type recordingProvider struct {
|
||||
providers.Provider
|
||||
updates []annotationUpdate
|
||||
}
|
||||
|
||||
func (p *recordingProvider) UpdateServiceAnnotation(_ context.Context, endpoint, endpointIPv6 string,
|
||||
_ *v1.Service, _ *kubernetes.Clientset) error {
|
||||
p.updates = append(p.updates, annotationUpdate{endpoint: endpoint, endpointIPv6: endpointIPv6})
|
||||
return nil
|
||||
}
|
||||
|
||||
func TestUpdateAnnotationsZeroEndpointsThenSameEndpoint(t *testing.T) {
|
||||
for _, enableEndpoints := range []bool{true, false} {
|
||||
providerName := "EndpointSlices"
|
||||
provider := providers.NewEndpointslices()
|
||||
if enableEndpoints {
|
||||
providerName = "Endpoints"
|
||||
provider = providers.NewEndpoints()
|
||||
}
|
||||
|
||||
for _, family := range []struct {
|
||||
name string
|
||||
endpoint string
|
||||
other string
|
||||
egressIPv6 bool
|
||||
}{
|
||||
{name: "IPv4", endpoint: "10.0.0.1", other: "fd00::1"},
|
||||
{name: "IPv6", endpoint: "fd00::1", other: "10.0.0.1", egressIPv6: true},
|
||||
} {
|
||||
t.Run(providerName+"/"+family.name, func(t *testing.T) {
|
||||
annotations := map[string]string{kubevip.Egress: "true"}
|
||||
if family.egressIPv6 {
|
||||
annotations[kubevip.EgressIPv6] = "true"
|
||||
}
|
||||
if !enableEndpoints {
|
||||
if family.egressIPv6 {
|
||||
annotations[kubevip.ActiveEndpoint] = family.other
|
||||
annotations[kubevip.ActiveEndpointIPv6] = family.endpoint
|
||||
} else {
|
||||
annotations[kubevip.ActiveEndpoint] = family.endpoint
|
||||
annotations[kubevip.ActiveEndpointIPv6] = family.other
|
||||
}
|
||||
} else {
|
||||
annotations[kubevip.ActiveEndpoint] = family.endpoint
|
||||
}
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "test-service", Namespace: "default", UID: "test-uid", Annotations: annotations,
|
||||
}}
|
||||
serviceInstance := &instance.Instance{ServiceUID: service.UID, ServiceSnapshot: service.DeepCopy()}
|
||||
instances := []*instance.Instance{serviceInstance}
|
||||
recorder := &recordingProvider{Provider: provider}
|
||||
processor := &Processor{
|
||||
config: &kubevip.Config{EnableEndpoints: enableEndpoints},
|
||||
provider: recorder,
|
||||
instances: &instances,
|
||||
}
|
||||
|
||||
noEndpoint := ""
|
||||
updated, changed := processor.updateAnnotations(service, serviceInstance, &noEndpoint, nil)
|
||||
if changed {
|
||||
serviceInstance.ServiceSnapshot = updated
|
||||
}
|
||||
repopulatedEndpoint := family.endpoint
|
||||
updated, changed = processor.updateAnnotations(service, serviceInstance, &repopulatedEndpoint, nil)
|
||||
if changed {
|
||||
serviceInstance.ServiceSnapshot = updated
|
||||
}
|
||||
|
||||
cleared := annotationUpdate{}
|
||||
repopulated := annotationUpdate{endpoint: family.endpoint}
|
||||
if !enableEndpoints {
|
||||
if family.egressIPv6 {
|
||||
cleared = annotationUpdate{endpoint: family.other}
|
||||
repopulated = annotationUpdate{endpoint: family.other, endpointIPv6: family.endpoint}
|
||||
} else {
|
||||
cleared = annotationUpdate{endpointIPv6: family.other}
|
||||
repopulated = annotationUpdate{endpoint: family.endpoint, endpointIPv6: family.other}
|
||||
}
|
||||
}
|
||||
want := []annotationUpdate{cleared, repopulated}
|
||||
if len(recorder.updates) != len(want) {
|
||||
t.Fatalf("annotation updates = %+v, want %+v", recorder.updates, want)
|
||||
}
|
||||
for index := range want {
|
||||
if recorder.updates[index] != want[index] {
|
||||
t.Errorf("annotation update %d = %+v, want %+v", index, recorder.updates[index], want[index])
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestUpdateAnnotationsEndpointSlicesClearsConfiguredFamily(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
egressIPv6 bool
|
||||
want annotationUpdate
|
||||
}{
|
||||
{name: "IPv4", want: annotationUpdate{endpointIPv6: "fd00::1"}},
|
||||
{name: "IPv6", egressIPv6: true, want: annotationUpdate{endpoint: "10.0.0.1"}},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
annotations := map[string]string{
|
||||
kubevip.Egress: "true",
|
||||
kubevip.ActiveEndpoint: "10.0.0.1",
|
||||
kubevip.ActiveEndpointIPv6: "fd00::1",
|
||||
}
|
||||
if test.egressIPv6 {
|
||||
annotations[kubevip.EgressIPv6] = "true"
|
||||
}
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "test-service", Namespace: "default", UID: "test-uid", Annotations: annotations,
|
||||
}}
|
||||
instances := []*instance.Instance{{ServiceUID: service.UID, ServiceSnapshot: service.DeepCopy()}}
|
||||
recorder := &recordingProvider{Provider: providers.NewEndpointslices()}
|
||||
processor := &Processor{
|
||||
config: &kubevip.Config{EnableEndpoints: false},
|
||||
provider: recorder,
|
||||
instances: &instances,
|
||||
}
|
||||
|
||||
noEndpoint := ""
|
||||
processor.updateAnnotations(service, instances[0], &noEndpoint, nil)
|
||||
|
||||
if len(recorder.updates) != 1 || recorder.updates[0] != test.want {
|
||||
t.Fatalf("annotation updates = %+v, want [%+v]", recorder.updates, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestUpdateAnnotationsValidatesEndpointFamily(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
endpoint string
|
||||
egressIPv6 bool
|
||||
want annotationUpdate
|
||||
wantUpdate bool
|
||||
}{
|
||||
{name: "invalid address", endpoint: "not-an-ip"},
|
||||
{name: "IPv6 endpoint for IPv4 egress", endpoint: "fd00::1"},
|
||||
{name: "IPv4 endpoint for IPv6 egress", endpoint: "10.0.0.1", egressIPv6: true},
|
||||
{name: "IPv4 endpoint", endpoint: "10.0.0.2", want: annotationUpdate{endpoint: "10.0.0.2", endpointIPv6: "fd00::1"}, wantUpdate: true},
|
||||
{name: "IPv6 endpoint", endpoint: "fd00::2", egressIPv6: true, want: annotationUpdate{endpoint: "10.0.0.1", endpointIPv6: "fd00::2"}, wantUpdate: true},
|
||||
}
|
||||
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
annotations := map[string]string{
|
||||
kubevip.Egress: "true",
|
||||
kubevip.ActiveEndpoint: "10.0.0.1",
|
||||
kubevip.ActiveEndpointIPv6: "fd00::1",
|
||||
}
|
||||
if test.egressIPv6 {
|
||||
annotations[kubevip.EgressIPv6] = "true"
|
||||
}
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "test-service", Namespace: "default", Annotations: annotations,
|
||||
}}
|
||||
recorder := &recordingProvider{Provider: providers.NewEndpointslices()}
|
||||
processor := &Processor{
|
||||
config: &kubevip.Config{EnableEndpoints: false},
|
||||
provider: recorder,
|
||||
}
|
||||
|
||||
processor.updateAnnotations(service, nil, &test.endpoint, nil)
|
||||
|
||||
if !test.wantUpdate {
|
||||
if len(recorder.updates) != 0 {
|
||||
t.Fatalf("annotation updates = %+v, want none", recorder.updates)
|
||||
}
|
||||
return
|
||||
}
|
||||
if len(recorder.updates) != 1 || recorder.updates[0] != test.want {
|
||||
t.Fatalf("annotation updates = %+v, want [%+v]", recorder.updates, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func (f *fakeWorker) processInstance(_ context.Context, _ *sync.Map, _ *v1.Service, _ *instance.Instance) error {
|
||||
if f.processHook != nil {
|
||||
f.processHook()
|
||||
}
|
||||
f.processCalled = true
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *fakeWorker) clear(_ *servicecontext.Context, _ *string, _ *v1.Service) {
|
||||
func (f *fakeWorker) clear(_ context.Context, _ *sync.Map, _ *string, _ *v1.Service, _ *instance.Instance) {
|
||||
f.clearCalled = true
|
||||
}
|
||||
|
||||
func (f *fakeWorker) getEndpoints(_ *v1.Service, _ string) ([]string, error) { return f.endpoints, nil }
|
||||
func (f *fakeWorker) removeEgress(_ *v1.Service, _ *string) {}
|
||||
func (f *fakeWorker) delete(_ context.Context, _ *v1.Service, _ string) error {
|
||||
return nil
|
||||
}
|
||||
func (f *fakeWorker) setInstanceEndpointsStatus(_ context.Context, _ *v1.Service, _ []string) error {
|
||||
func (f *fakeWorker) setInstanceEndpointsStatus(_ *v1.Service, _ *instance.Instance, _ []string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
func TestAddOrModify_ZeroEndpointsBehavior(t *testing.T) {
|
||||
func noOpServiceLock(types.UID) func() {
|
||||
return func() {}
|
||||
}
|
||||
|
||||
// TestReconcile_RecomputesRemainingEndpoints asserts that deleting one EndpointSlice
|
||||
// reconciles against the endpoints that remain, instead of assuming the service
|
||||
// lost all of them.
|
||||
func TestReconcile_RecomputesRemainingEndpoints(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
remaining []string
|
||||
lastKnown string
|
||||
expectReady bool
|
||||
expectClear bool
|
||||
expectProcess bool
|
||||
expectedLastKnown string
|
||||
}{
|
||||
{
|
||||
name: "remaining endpoints keep the service up",
|
||||
remaining: []string{"10.0.0.2"},
|
||||
lastKnown: "10.0.0.2",
|
||||
expectReady: true,
|
||||
expectProcess: true,
|
||||
expectedLastKnown: "10.0.0.2",
|
||||
},
|
||||
{
|
||||
name: "stale last known endpoint moves to a survivor",
|
||||
remaining: []string{"10.0.0.2"},
|
||||
lastKnown: "10.0.0.1",
|
||||
expectReady: true,
|
||||
expectProcess: true,
|
||||
expectedLastKnown: "10.0.0.2",
|
||||
},
|
||||
{
|
||||
name: "last endpoint removed tears the service down",
|
||||
remaining: nil,
|
||||
lastKnown: "10.0.0.1",
|
||||
expectReady: false,
|
||||
expectClear: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
worker := &fakeWorker{endpoints: test.remaining}
|
||||
p := &Processor{
|
||||
config: &kubevip.Config{},
|
||||
provider: providers.NewEndpointslices(),
|
||||
worker: worker,
|
||||
lockService: noOpServiceLock,
|
||||
}
|
||||
|
||||
svcCtx := servicecontext.New(context.Background())
|
||||
svcCtx.SignalReadiness()
|
||||
|
||||
lastKnown := test.lastKnown
|
||||
restart, err := p.Reconcile(
|
||||
svcCtx,
|
||||
watch.Event{
|
||||
Type: watch.Deleted,
|
||||
Object: &discoveryv1.EndpointSlice{ObjectMeta: metav1.ObjectMeta{Name: "slice-1"}},
|
||||
},
|
||||
&lastKnown,
|
||||
&v1.Service{Spec: v1.ServiceSpec{ExternalTrafficPolicy: v1.ServiceExternalTrafficPolicyTypeLocal}},
|
||||
"node-1",
|
||||
&sync.WaitGroup{},
|
||||
nil,
|
||||
nil,
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("Reconcile returned error: %v", err)
|
||||
}
|
||||
if restart {
|
||||
t.Fatal("Reconcile unexpectedly requested restart")
|
||||
}
|
||||
|
||||
if ready := svcCtx.IsReady(); ready != test.expectReady {
|
||||
t.Fatalf("readiness mismatch: expected %v, got %v", test.expectReady, ready)
|
||||
}
|
||||
if worker.clearCalled != test.expectClear {
|
||||
t.Fatalf("clearCalled mismatch: expected %v, got %v", test.expectClear, worker.clearCalled)
|
||||
}
|
||||
if worker.processCalled != test.expectProcess {
|
||||
t.Fatalf("processCalled mismatch: expected %v, got %v", test.expectProcess, worker.processCalled)
|
||||
}
|
||||
if test.expectedLastKnown != "" && lastKnown != test.expectedLastKnown {
|
||||
t.Fatalf("lastKnownGoodEndpoint mismatch: expected %q, got %q", test.expectedLastKnown, lastKnown)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestReconcile_ZeroEndpointsBehavior(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
run := func(t *testing.T, service *v1.Service, presetSignalled bool, expectReady bool, expectClear bool, expectProcess bool) {
|
||||
@@ -68,9 +355,10 @@ func TestAddOrModify_ZeroEndpointsBehavior(t *testing.T) {
|
||||
|
||||
worker := &fakeWorker{endpoints: []string{}}
|
||||
p := &Processor{
|
||||
config: &kubevip.Config{},
|
||||
provider: providers.NewEndpointslices(),
|
||||
worker: worker,
|
||||
config: &kubevip.Config{},
|
||||
provider: providers.NewEndpointslices(),
|
||||
worker: worker,
|
||||
lockService: noOpServiceLock,
|
||||
}
|
||||
|
||||
svcCtx := servicecontext.New(context.Background())
|
||||
@@ -78,25 +366,24 @@ func TestAddOrModify_ZeroEndpointsBehavior(t *testing.T) {
|
||||
svcCtx.SignalReadiness()
|
||||
}
|
||||
|
||||
restart, err := p.AddOrModify(
|
||||
restart, err := p.Reconcile(
|
||||
svcCtx,
|
||||
watch.Event{Type: watch.Modified, Object: &discoveryv1.EndpointSlice{}},
|
||||
new(string),
|
||||
service,
|
||||
"node-1",
|
||||
func(*servicecontext.Context, *v1.Service, *sync.WaitGroup, bool) error { return nil },
|
||||
&sync.WaitGroup{},
|
||||
nil,
|
||||
nil,
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("AddOrModify returned error: %v", err)
|
||||
t.Fatalf("Reconcile returned error: %v", err)
|
||||
}
|
||||
if restart {
|
||||
t.Fatal("AddOrModify unexpectedly requested restart")
|
||||
t.Fatal("Reconcile unexpectedly requested restart")
|
||||
}
|
||||
|
||||
if ready := svcCtx.Signalled.Load(); ready != expectReady {
|
||||
if ready := svcCtx.IsReady(); ready != expectReady {
|
||||
t.Fatalf("readiness mismatch: expected %v, got %v", expectReady, ready)
|
||||
}
|
||||
if worker.clearCalled != expectClear {
|
||||
@@ -131,3 +418,162 @@ func TestAddOrModify_ZeroEndpointsBehavior(t *testing.T) {
|
||||
run(t, service, true, false, true, false)
|
||||
})
|
||||
}
|
||||
|
||||
func TestHandleNoEndpointsStopsGlobalRoutingTableWorkers(t *testing.T) {
|
||||
config := &kubevip.Config{EnableRoutingTable: true, KubernetesLeaderElection: kubevip.KubernetesLeaderElection{EnableLeaderElection: true}}
|
||||
serviceCluster, err := cluster.InitCluster(&kubevip.Config{}, true, nil, nil, route.NewManager(), nil)
|
||||
if err != nil {
|
||||
t.Fatalf("initializing Service cluster: %v", err)
|
||||
}
|
||||
var workers sync.WaitGroup
|
||||
if err := serviceCluster.StartLoadBalancerService(context.Background(), config, nil, "service", &workers); err != nil {
|
||||
t.Fatalf("starting Service cluster: %v", err)
|
||||
}
|
||||
instance := &instance.Instance{Clusters: []*cluster.Cluster{serviceCluster}}
|
||||
processor := &Processor{config: config, worker: &fakeWorker{}}
|
||||
|
||||
processor.handleNoEndpoints(servicecontext.New(context.Background()), &v1.Service{}, instance, new(string))
|
||||
if err := serviceCluster.StartLoadBalancerService(context.Background(), config, nil, "service", &workers); err != nil {
|
||||
t.Fatalf("starting Service cluster after endpoint loss: %v", err)
|
||||
}
|
||||
serviceCluster.StopAndWait()
|
||||
workers.Wait()
|
||||
}
|
||||
|
||||
func TestReconcileIPv6EgressWithoutIPv6EndpointsClearsReadiness(t *testing.T) {
|
||||
worker := &fakeWorker{endpoints: []string{"192.0.2.10"}}
|
||||
processor := &Processor{
|
||||
config: &kubevip.Config{},
|
||||
provider: providers.NewEndpointslices(),
|
||||
worker: worker,
|
||||
lockService: noOpServiceLock,
|
||||
}
|
||||
service := &v1.Service{
|
||||
ObjectMeta: metav1.ObjectMeta{Annotations: map[string]string{kubevip.EgressIPv6: "true"}},
|
||||
Spec: v1.ServiceSpec{ExternalTrafficPolicy: v1.ServiceExternalTrafficPolicyTypeLocal},
|
||||
}
|
||||
svcCtx := servicecontext.New(context.Background())
|
||||
svcCtx.SignalReadiness()
|
||||
|
||||
restart, err := processor.Reconcile(svcCtx, watch.Event{Type: watch.Modified, Object: &discoveryv1.EndpointSlice{}},
|
||||
new(string), service, "node", &sync.WaitGroup{}, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("Reconcile() error = %v", err)
|
||||
}
|
||||
if restart {
|
||||
t.Fatal("Reconcile requested a restart for an unusable endpoint family")
|
||||
}
|
||||
if svcCtx.IsReady() {
|
||||
t.Fatal("IPv6 egress remained ready with only IPv4 endpoints")
|
||||
}
|
||||
if !worker.clearCalled {
|
||||
t.Fatal("IPv6 egress did not clear the worker with only IPv4 endpoints")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSharedServiceVIPRequiresClusterPolicyAndElectionLease(t *testing.T) {
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "first", Namespace: "default", Annotations: map[string]string{kubevip.ServiceLease: "shared"},
|
||||
}, Spec: v1.ServiceSpec{LoadBalancerIP: "192.0.2.10", ExternalTrafficPolicy: v1.ServiceExternalTrafficPolicyTypeCluster}}
|
||||
candidate := service.DeepCopy()
|
||||
candidate.Name = "second"
|
||||
first := &instance.Instance{ServiceSnapshot: service}
|
||||
second := &instance.Instance{ServiceSnapshot: candidate}
|
||||
|
||||
if len(sharedServiceVIPs(&kubevip.Config{EnableServicesElection: true}, first, []*instance.Instance{second})) == 0 {
|
||||
t.Fatal("shared lease Services with Cluster traffic policy did not share their VIP")
|
||||
}
|
||||
if len(sharedServiceVIPs(&kubevip.Config{}, first, []*instance.Instance{second})) == 0 {
|
||||
t.Fatal("Cluster policy Services without per-Service election did not share their VIP")
|
||||
}
|
||||
candidate.Spec.ExternalTrafficPolicy = v1.ServiceExternalTrafficPolicyTypeLocal
|
||||
if len(sharedServiceVIPs(&kubevip.Config{}, first, []*instance.Instance{second})) != 0 {
|
||||
t.Fatal("Local traffic policy Services shared a VIP")
|
||||
}
|
||||
candidate.Spec.ExternalTrafficPolicy = v1.ServiceExternalTrafficPolicyTypeCluster
|
||||
candidate.Annotations[kubevip.ServiceLease] = "different"
|
||||
if len(sharedServiceVIPs(&kubevip.Config{EnableServicesElection: true}, first, []*instance.Instance{second})) != 0 {
|
||||
t.Fatal("Services with different leases shared a VIP")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSharedServiceVIPsReturnsOnlyOverlappingAddresses(t *testing.T) {
|
||||
first := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "first", Namespace: "default", Annotations: map[string]string{kubevip.ServiceLease: "shared"},
|
||||
}, Spec: v1.ServiceSpec{LoadBalancerIP: "192.0.2.10", ExternalTrafficPolicy: v1.ServiceExternalTrafficPolicyTypeCluster}}
|
||||
second := first.DeepCopy()
|
||||
second.Name = "second"
|
||||
second.Spec.LoadBalancerIP = "192.0.2.11"
|
||||
firstInstance := &instance.Instance{ServiceSnapshot: first, ServiceAddresses: []string{"192.0.2.10", "2001:db8::10"}}
|
||||
secondInstance := &instance.Instance{ServiceSnapshot: second, ServiceAddresses: []string{"192.0.2.10", "2001:db8::11"}}
|
||||
|
||||
shared := sharedServiceVIPs(&kubevip.Config{EnableServicesElection: true}, firstInstance, []*instance.Instance{secondInstance})
|
||||
if len(shared) != 1 {
|
||||
t.Fatalf("shared VIPs = %v, want exactly one", shared)
|
||||
}
|
||||
if _, found := shared["192.0.2.10"]; !found {
|
||||
t.Fatalf("shared VIPs = %v, missing overlapping IPv4 address", shared)
|
||||
}
|
||||
}
|
||||
|
||||
// TestReconcileServicesElectionDoesNotStartElectionLoop asserts endpoint events
|
||||
// only update readiness. The services coordinator owns the election loop.
|
||||
func TestReconcileServicesElectionDoesNotStartElectionLoop(t *testing.T) {
|
||||
config := &kubevip.Config{
|
||||
EnableServicesElection: true,
|
||||
LeaderElectionType: "kubernetes",
|
||||
}
|
||||
|
||||
service := &v1.Service{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "test-svc", Namespace: "default", UID: "test-uid"},
|
||||
Spec: v1.ServiceSpec{Type: v1.ServiceTypeLoadBalancer},
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
leaseMgr := lease.NewManager()
|
||||
leaseNamespace, serviceLease := lease.ServiceName(service)
|
||||
svcLease := leaseMgr.Add(ctx, lease.NewID(config.LeaderElectionType, leaseNamespace, serviceLease))
|
||||
|
||||
svcCtx := servicecontext.New(svcLease.Ctx)
|
||||
|
||||
wg := &sync.WaitGroup{}
|
||||
defer wg.Wait()
|
||||
defer svcCtx.Cancel()
|
||||
|
||||
p := &Processor{
|
||||
config: config,
|
||||
provider: providers.NewEndpointslices(),
|
||||
worker: &fakeWorker{endpoints: []string{"10.0.0.1"}},
|
||||
leaseMgr: leaseMgr,
|
||||
lockService: noOpServiceLock,
|
||||
}
|
||||
|
||||
// Three endpoint events, as a flapping backend pod would produce.
|
||||
for range 3 {
|
||||
restart, err := p.Reconcile(svcCtx, watch.Event{Type: watch.Modified, Object: &discoveryv1.EndpointSlice{}},
|
||||
new(string), service, "node-1", wg, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("Reconcile returned error: %v", err)
|
||||
}
|
||||
if restart {
|
||||
t.Fatal("Reconcile unexpectedly requested restart")
|
||||
}
|
||||
}
|
||||
|
||||
generation, ready, lost, isReady := svcCtx.ReadinessState()
|
||||
if generation != 1 || !isReady {
|
||||
t.Fatalf("readiness state = generation %d, ready %t; want generation 1 ready", generation, isReady)
|
||||
}
|
||||
select {
|
||||
case <-ready:
|
||||
default:
|
||||
t.Fatal("endpoint reconciliation did not signal readiness")
|
||||
}
|
||||
select {
|
||||
case <-lost:
|
||||
t.Fatal("endpoint reconciliation unexpectedly reset readiness")
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,16 +3,14 @@ package endpoints
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"sync"
|
||||
|
||||
log "log/slog"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/bgp"
|
||||
"github.com/kube-vip/kube-vip/pkg/endpoints/providers"
|
||||
"github.com/kube-vip/kube-vip/pkg/instance"
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
"github.com/kube-vip/kube-vip/pkg/nftables"
|
||||
"github.com/kube-vip/kube-vip/pkg/servicecontext"
|
||||
"github.com/kube-vip/kube-vip/pkg/utils"
|
||||
"github.com/kube-vip/kube-vip/pkg/wireguard"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
@@ -22,27 +20,20 @@ import (
|
||||
type wireguardWorker struct {
|
||||
config *kubevip.Config
|
||||
provider providers.Provider
|
||||
bgpServer *bgp.Server
|
||||
instances *[]*instance.Instance
|
||||
leaseMgr *lease.Manager
|
||||
tunnelMgr *wireguard.TunnelManager
|
||||
}
|
||||
|
||||
func newWireguardWorker(config *kubevip.Config, provider providers.Provider, bgpServer *bgp.Server,
|
||||
instances *[]*instance.Instance, leaseMgr *lease.Manager, tunnelMgr *wireguard.TunnelManager) *wireguardWorker {
|
||||
func newWireguardWorker(config *kubevip.Config, provider providers.Provider, tunnelMgr *wireguard.TunnelManager) *wireguardWorker {
|
||||
return &wireguardWorker{
|
||||
config: config,
|
||||
provider: provider,
|
||||
bgpServer: bgpServer,
|
||||
instances: instances,
|
||||
leaseMgr: leaseMgr,
|
||||
tunnelMgr: tunnelMgr,
|
||||
}
|
||||
}
|
||||
|
||||
// processInstance updates nftables DNAT rules when endpoints change
|
||||
// This is called by the endpoint watcher when endpoints are added/modified
|
||||
func (w *wireguardWorker) processInstance(svcCtx *servicecontext.Context, service *v1.Service) error {
|
||||
func (w *wireguardWorker) processInstance(ctx context.Context, _ *sync.Map, service *v1.Service, inst *instance.Instance) error {
|
||||
log.Debug("[wireguard] processing instance for endpoint change", "service", service.Name, "namespace", service.Namespace)
|
||||
|
||||
// Get the target endpoint for this service
|
||||
@@ -61,25 +52,21 @@ func (w *wireguardWorker) processInstance(svcCtx *servicecontext.Context, servic
|
||||
|
||||
if len(endpoints) == 0 {
|
||||
log.Debug("[wireguard] no endpoints available", "service", service.Name)
|
||||
w.clear(svcCtx, nil, service)
|
||||
w.clear(ctx, nil, nil, service, inst)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Find the service processor to call updateServiceWireguardEndpoints
|
||||
// Note: This requires access to the service processor which we don't have here
|
||||
// So we'll recreate the DNAT rules directly
|
||||
|
||||
// First, clear existing rules
|
||||
w.clear(svcCtx, nil, service)
|
||||
|
||||
// Get service VIPs
|
||||
serviceIPs, err := utils.FetchServiceIPs(service)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to get service IPs: %w", err)
|
||||
}
|
||||
|
||||
// Create service identifier
|
||||
serviceID := utils.SanitizeServiceID(fmt.Sprintf("%s_%s", service.Namespace, service.Name))
|
||||
if len(service.Spec.Ports) != 0 {
|
||||
if err := w.ensureTunnels(service, serviceIPs); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
w.clearDNAT(service)
|
||||
|
||||
log.Info("[wireguard] updating DNAT rules for endpoint change",
|
||||
"service", service.Name,
|
||||
@@ -87,133 +74,124 @@ func (w *wireguardWorker) processInstance(svcCtx *servicecontext.Context, servic
|
||||
"endpoints", endpoints,
|
||||
"vips", serviceIPs)
|
||||
|
||||
// Update DNAT rules for each port
|
||||
for _, port := range service.Spec.Ports {
|
||||
// Determine target port (resolve named ports if necessary)
|
||||
if port.Protocol != v1.ProtocolTCP && port.Protocol != v1.ProtocolUDP {
|
||||
continue
|
||||
}
|
||||
targetPort := w.provider.ResolvePort(port)
|
||||
log.Info("[wireguard] resolved port", "service", service.Name, "servicePort", port.Port, "targetPort", targetPort, "targetPortName", port.TargetPort.StrVal)
|
||||
|
||||
// Build targets list from all endpoints
|
||||
targets := make([]nftables.DNATTarget, len(endpoints))
|
||||
for i, ep := range endpoints {
|
||||
targets[i] = nftables.DNATTarget{
|
||||
IP: ep,
|
||||
for index, endpoint := range endpoints {
|
||||
targets[index] = nftables.DNATTarget{
|
||||
IP: endpoint,
|
||||
Port: uint16(targetPort), //nolint:gosec // Port range validated by Kubernetes
|
||||
}
|
||||
}
|
||||
portServiceID, _ := wireguard.ServicePortIDs(service.Namespace, service.Name, port)
|
||||
|
||||
for _, vip := range serviceIPs {
|
||||
// Strip CIDR notation if present
|
||||
vipAddr := utils.StripCIDR(vip)
|
||||
|
||||
// Get WireGuard interface name from TunnelManager for this VIP
|
||||
for _, serviceIP := range serviceIPs {
|
||||
vipAddress := utils.StripCIDR(serviceIP)
|
||||
if w.tunnelMgr == nil {
|
||||
log.Error("[wireguard] TunnelManager not configured; cannot update DNAT rules",
|
||||
"service", service.Name,
|
||||
"namespace", service.Namespace)
|
||||
return fmt.Errorf("TunnelManager not configured")
|
||||
return fmt.Errorf("WireGuard tunnel manager not configured")
|
||||
}
|
||||
tunnelConfig := w.tunnelMgr.GetConfigForVIP(vipAddr)
|
||||
tunnelConfig := w.tunnelMgr.GetConfigForVIP(vipAddress)
|
||||
if tunnelConfig == nil {
|
||||
log.Error("[wireguard] WireGuard interface name not configured; cannot update DNAT rules",
|
||||
"service", service.Name,
|
||||
"namespace", service.Namespace,
|
||||
"vip", vipAddr)
|
||||
return fmt.Errorf("wireguard interface name not configured for VIP %s", vipAddr)
|
||||
return fmt.Errorf("wireguard interface name not configured for VIP %s", vipAddress)
|
||||
}
|
||||
wgInterface := tunnelConfig.InterfaceName
|
||||
|
||||
portServiceID := fmt.Sprintf("%s_p%d", serviceID, port.Port)
|
||||
|
||||
log.Info("[wireguard] applying DNAT rule with load balancing",
|
||||
"service", service.Name,
|
||||
"vip", vipAddr,
|
||||
"interface", wgInterface,
|
||||
"sourcePort", port.Port,
|
||||
"targets", targets,
|
||||
"chainID", portServiceID)
|
||||
|
||||
// Apply the DNAT rule with load balancing across all endpoints
|
||||
// localEndpoint=true when using ExternalTrafficPolicy=Local, which preserves client source IP
|
||||
isLocalEndpoint := service.Spec.ExternalTrafficPolicy == v1.ServiceExternalTrafficPolicyTypeLocal
|
||||
err := nftables.ApplyDNAT(
|
||||
wgInterface,
|
||||
vipAddr,
|
||||
if err := nftables.ApplyDNAT(
|
||||
tunnelConfig.InterfaceName,
|
||||
vipAddress,
|
||||
uint16(port.Port), //nolint:gosec // Port range validated by Kubernetes
|
||||
targets,
|
||||
portServiceID,
|
||||
port.Protocol,
|
||||
isLocalEndpoint,
|
||||
service.Spec.ExternalTrafficPolicy == v1.ServiceExternalTrafficPolicyTypeLocal,
|
||||
tunnelConfig.ListenPort,
|
||||
)
|
||||
if err != nil {
|
||||
log.Error("[wireguard] failed to update DNAT rule",
|
||||
"service", service.Name,
|
||||
"vip", vipAddr,
|
||||
"port", port.Port,
|
||||
"err", err)
|
||||
); err != nil {
|
||||
log.Error("[wireguard] failed to update DNAT rule", "service", service.Name, "vip", vipAddress, "port", port.Port, "err", err)
|
||||
continue
|
||||
}
|
||||
|
||||
log.Debug("[wireguard] DNAT rule updated successfully",
|
||||
"service", service.Name,
|
||||
"vip", vipAddr,
|
||||
"port", port.Port,
|
||||
"targetCount", len(targets))
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (w *wireguardWorker) ensureTunnels(service *v1.Service, serviceIPs []string) error {
|
||||
if w.tunnelMgr == nil {
|
||||
return fmt.Errorf("WireGuard tunnel manager not configured")
|
||||
}
|
||||
if len(serviceIPs) == 0 {
|
||||
return fmt.Errorf("no service IPs found for service %s/%s", service.Namespace, service.Name)
|
||||
}
|
||||
var successCount int
|
||||
var lastErr error
|
||||
for _, serviceIP := range serviceIPs {
|
||||
if !w.tunnelMgr.HasConfigForVIP(serviceIP) {
|
||||
lastErr = fmt.Errorf("no WireGuard tunnel configuration found for VIP %s", serviceIP)
|
||||
continue
|
||||
}
|
||||
if err := w.tunnelMgr.AcquireTunnelForVIP(serviceIP, string(service.UID)); err != nil {
|
||||
lastErr = fmt.Errorf("bring up WireGuard tunnel for VIP %s: %w", serviceIP, err)
|
||||
continue
|
||||
}
|
||||
successCount++
|
||||
}
|
||||
if successCount == 0 {
|
||||
return fmt.Errorf("failed to setup WireGuard tunnel for any VIP in service %s/%s: %w", service.Namespace, service.Name, lastErr)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// clear removes DNAT rules when no endpoints are available
|
||||
func (w *wireguardWorker) clear(svcCtx *servicecontext.Context, lastKnownGoodEndpoint *string, service *v1.Service) {
|
||||
func (w *wireguardWorker) clear(_ context.Context, _ *sync.Map, _ *string, service *v1.Service, _ *instance.Instance) {
|
||||
w.clearDNAT(service)
|
||||
}
|
||||
|
||||
func (w *wireguardWorker) clearDNAT(service *v1.Service) {
|
||||
log.Info("[wireguard] clearing DNAT rules (no endpoints)", "service", service.Name, "namespace", service.Namespace)
|
||||
forEachServiceDNATChain(service, func(ipv6 bool, serviceID string) {
|
||||
if err := nftables.DeleteIngressChains(ipv6, serviceID); err != nil {
|
||||
family := utils.IPv4Family
|
||||
if ipv6 {
|
||||
family = utils.IPv6Family
|
||||
}
|
||||
log.Warn("[wireguard] failed to delete DNAT chains", "family", family, "service", service.Name, "err", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
serviceID := utils.SanitizeServiceID(fmt.Sprintf("%s_%s", service.Namespace, service.Name))
|
||||
|
||||
// Get service IPs to determine IPv4 vs IPv6
|
||||
func forEachServiceDNATChain(service *v1.Service, visit func(bool, string)) {
|
||||
if service == nil {
|
||||
return
|
||||
}
|
||||
serviceIPs, _ := utils.FetchServiceIPs(service)
|
||||
familyUnknown := len(serviceIPs) == 0
|
||||
hasIPv4, hasIPv6 := familyUnknown, familyUnknown
|
||||
for _, serviceIP := range serviceIPs {
|
||||
if isIPv6Address(serviceIP) {
|
||||
hasIPv6 = true
|
||||
} else {
|
||||
hasIPv4 = true
|
||||
}
|
||||
}
|
||||
|
||||
// Delete DNAT chains for each port
|
||||
for _, port := range service.Spec.Ports {
|
||||
if port.Protocol != v1.ProtocolTCP && port.Protocol != v1.ProtocolUDP {
|
||||
continue
|
||||
}
|
||||
|
||||
portServiceID := fmt.Sprintf("%s_p%d", serviceID, port.Port)
|
||||
|
||||
// Determine if we have IPv4 or IPv6
|
||||
hasIPv4, hasIPv6 := false, false
|
||||
for _, vip := range serviceIPs {
|
||||
if isIPv6Address(vip) {
|
||||
hasIPv6 = true
|
||||
} else {
|
||||
hasIPv4 = true
|
||||
portServiceID, legacyServiceID := wireguard.ServicePortIDs(service.Namespace, service.Name, port)
|
||||
// The legacy identifier is visited so an upgrade removes chains written
|
||||
// before rule IDs carried the protocol.
|
||||
for _, serviceID := range []string{portServiceID, legacyServiceID} {
|
||||
if hasIPv4 {
|
||||
visit(false, serviceID)
|
||||
}
|
||||
if hasIPv6 {
|
||||
visit(true, serviceID)
|
||||
}
|
||||
}
|
||||
|
||||
if hasIPv4 {
|
||||
if err := nftables.DeleteIngressChains(false, portServiceID); err != nil {
|
||||
log.Warn("[wireguard] failed to delete IPv4 DNAT chains",
|
||||
"service", service.Name,
|
||||
"port", port.Port,
|
||||
"err", err)
|
||||
}
|
||||
}
|
||||
|
||||
if hasIPv6 {
|
||||
if err := nftables.DeleteIngressChains(true, portServiceID); err != nil {
|
||||
log.Warn("[wireguard] failed to delete IPv6 DNAT chains",
|
||||
"service", service.Name,
|
||||
"port", port.Port,
|
||||
"err", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if svcCtx.LeaderCancel != nil {
|
||||
svcCtx.LeaderCancel()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -242,16 +220,8 @@ func (w *wireguardWorker) removeEgress(service *v1.Service, lastKnownGoodEndpoin
|
||||
log.Debug("[wireguard] removeEgress called (no-op)", "service", service.Name)
|
||||
}
|
||||
|
||||
// delete removes all DNAT rules for a service
|
||||
func (w *wireguardWorker) delete(ctx context.Context, service *v1.Service, id string) error {
|
||||
log.Info("[wireguard] deleting DNAT rules for service", "service", service.Name, "namespace", service.Namespace)
|
||||
|
||||
w.clear(nil, nil, service)
|
||||
return nil
|
||||
}
|
||||
|
||||
// setInstanceEndpointsStatus updates the endpoint status on the service instance
|
||||
func (w *wireguardWorker) setInstanceEndpointsStatus(_ context.Context, service *v1.Service, endpoints []string) error {
|
||||
func (w *wireguardWorker) setInstanceEndpointsStatus(service *v1.Service, inst *instance.Instance, endpoints []string) error {
|
||||
hasEndpoints := len(endpoints) > 0
|
||||
|
||||
log.Debug("[wireguard] setting instance endpoint status",
|
||||
@@ -259,23 +229,17 @@ func (w *wireguardWorker) setInstanceEndpointsStatus(_ context.Context, service
|
||||
"hasEndpoints", hasEndpoints,
|
||||
"endpointCount", len(endpoints))
|
||||
|
||||
// Find the service instance
|
||||
for _, inst := range *w.instances {
|
||||
if inst.ServiceSnapshot == nil {
|
||||
continue
|
||||
}
|
||||
if inst.ServiceSnapshot.UID == service.UID {
|
||||
// Update the network status for all clusters
|
||||
for _, cluster := range inst.Clusters {
|
||||
for i := range cluster.Network {
|
||||
cluster.Network[i].SetHasEndpoints(hasEndpoints)
|
||||
}
|
||||
if inst != nil {
|
||||
// Update the network status for all clusters
|
||||
for _, cluster := range inst.Clusters {
|
||||
for i := range cluster.Network {
|
||||
cluster.Network[i].SetHasEndpoints(hasEndpoints)
|
||||
}
|
||||
log.Debug("[wireguard] updated instance endpoint status",
|
||||
"service", service.Name,
|
||||
"hasEndpoints", hasEndpoints)
|
||||
return nil
|
||||
}
|
||||
log.Debug("[wireguard] updated instance endpoint status",
|
||||
"service", service.Name,
|
||||
"hasEndpoints", hasEndpoints)
|
||||
return nil
|
||||
}
|
||||
|
||||
log.Debug("[wireguard] instance not found for endpoint status update", "service", service.Name)
|
||||
|
||||
41
pkg/endpoints/endpoints_wireguard_test.go
Normal file
41
pkg/endpoints/endpoints_wireguard_test.go
Normal file
@@ -0,0 +1,41 @@
|
||||
package endpoints
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"testing"
|
||||
|
||||
v1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
)
|
||||
|
||||
type recordingTunnelReleaser struct {
|
||||
releases []string
|
||||
}
|
||||
|
||||
func (r *recordingTunnelReleaser) ReleaseTunnelForVIP(vip, owner string) error {
|
||||
r.releases = append(r.releases, fmt.Sprintf("%s:%s", vip, owner))
|
||||
return nil
|
||||
}
|
||||
|
||||
func TestWireguardClearDoesNotDereferenceNilServiceContext(t *testing.T) {
|
||||
worker := &wireguardWorker{}
|
||||
service := &v1.Service{}
|
||||
|
||||
worker.clear(context.TODO(), nil, nil, service, nil)
|
||||
}
|
||||
|
||||
func TestReleaseWireguardServiceTunnelsUsesServiceUIDOwner(t *testing.T) {
|
||||
releaser := &recordingTunnelReleaser{}
|
||||
service := &v1.Service{
|
||||
ObjectMeta: metav1.ObjectMeta{UID: types.UID("service-uid")},
|
||||
Spec: v1.ServiceSpec{LoadBalancerIP: "192.0.2.10"},
|
||||
}
|
||||
|
||||
releaseWireguardServiceTunnels(releaser, service)
|
||||
|
||||
if len(releaser.releases) != 1 || releaser.releases[0] != "192.0.2.10:service-uid" {
|
||||
t.Fatalf("tunnel releases = %v, want [192.0.2.10:service-uid]", releaser.releases)
|
||||
}
|
||||
}
|
||||
@@ -61,6 +61,18 @@ func (ep *Endpoints) LoadObject(endpoints runtime.Object, cancel context.CancelF
|
||||
return nil
|
||||
}
|
||||
|
||||
// DeleteObject drops the tracked object. A service is backed by exactly one
|
||||
// v1.Endpoints object, so there is nothing to match on and the cache is reset.
|
||||
func (ep *Endpoints) DeleteObject(endpoints runtime.Object) error {
|
||||
//nolint:staticcheck // SA1019 endpoints have to be explicitly requested now
|
||||
if _, ok := endpoints.(*v1.Endpoints); !ok {
|
||||
return fmt.Errorf("[%s] unable to parse Kubernetes object", ep.GetLabel())
|
||||
}
|
||||
//nolint:staticcheck // SA1019 endpoints have to be explicitly requested now
|
||||
ep.endpoints = &v1.Endpoints{}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (ep *Endpoints) GetAllEndpoints() ([]string, error) {
|
||||
result := []string{}
|
||||
for subset := range ep.endpoints.Subsets {
|
||||
@@ -87,7 +99,7 @@ func (ep *Endpoints) GetLocalEndpoints(id string, _ *kubevip.Config) ([]string,
|
||||
continue
|
||||
}
|
||||
// 2. Compare the Hostname (only useful if address.NodeName is not available)
|
||||
if id == address.Hostname {
|
||||
if address.NodeName == nil && id == address.Hostname {
|
||||
log.Debug("found local endpoint", "label", ep.label, "ip", address.IP, "hostname", address.Hostname)
|
||||
localEndpoints = append(localEndpoints, address.IP)
|
||||
continue
|
||||
|
||||
@@ -20,15 +20,14 @@ import (
|
||||
)
|
||||
|
||||
type Endpointslices struct {
|
||||
label string
|
||||
endpointsv4 []discoveryv1.Endpoint
|
||||
endpointsv6 []discoveryv1.Endpoint
|
||||
ports []discoveryv1.EndpointPort
|
||||
label string
|
||||
slices map[string]*discoveryv1.EndpointSlice
|
||||
}
|
||||
|
||||
func NewEndpointslices() Provider {
|
||||
return &Endpointslices{
|
||||
label: "endpointslices",
|
||||
label: "endpointslices",
|
||||
slices: make(map[string]*discoveryv1.EndpointSlice),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -59,56 +58,71 @@ func (ep *Endpointslices) LoadObject(endpoints runtime.Object, cancel context.Ca
|
||||
return fmt.Errorf("[%s] error casting endpoints to v1.Endpoints struct", ep.label)
|
||||
}
|
||||
|
||||
if eps.AddressType == discoveryv1.AddressTypeIPv6 {
|
||||
ep.endpointsv6 = eps.Endpoints
|
||||
} else {
|
||||
ep.endpointsv4 = eps.Endpoints
|
||||
if ep.slices == nil {
|
||||
ep.slices = make(map[string]*discoveryv1.EndpointSlice)
|
||||
}
|
||||
|
||||
// Store ports for resolving named ports
|
||||
ep.ports = eps.Ports
|
||||
ep.slices[eps.Name] = eps.DeepCopy()
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (ep *Endpointslices) DeleteObject(endpoints runtime.Object) error {
|
||||
eps, ok := endpoints.(*discoveryv1.EndpointSlice)
|
||||
if !ok {
|
||||
return fmt.Errorf("[%s] unable to parse Kubernetes object", ep.GetLabel())
|
||||
}
|
||||
delete(ep.slices, eps.Name)
|
||||
return nil
|
||||
}
|
||||
|
||||
// isServing reports whether an endpoint should receive traffic. Per the
|
||||
// EndpointConditions godoc a nil Serving defers to Ready, and a nil Ready is an
|
||||
// unknown state that consumers should interpret as ready.
|
||||
func isServing(conditions discoveryv1.EndpointConditions) bool {
|
||||
serving := conditions.Serving
|
||||
if serving == nil {
|
||||
serving = conditions.Ready
|
||||
}
|
||||
return serving == nil || *serving
|
||||
}
|
||||
|
||||
func (ep *Endpointslices) GetAllEndpoints() ([]string, error) {
|
||||
result := []string{}
|
||||
for _, e := range ep.endpointsv4 {
|
||||
result = append(result, e.Addresses...)
|
||||
}
|
||||
for _, e := range ep.endpointsv6 {
|
||||
result = append(result, e.Addresses...)
|
||||
for _, eps := range ep.slices {
|
||||
for _, e := range eps.Endpoints {
|
||||
if !isServing(e.Conditions) {
|
||||
continue
|
||||
}
|
||||
result = append(result, e.Addresses...)
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func (ep *Endpointslices) GetLocalEndpoints(id string, _ *kubevip.Config) ([]string, error) {
|
||||
var localEndpoints []string
|
||||
tmpEps := []discoveryv1.Endpoint{}
|
||||
|
||||
tmpEps = append(tmpEps, ep.endpointsv4...)
|
||||
tmpEps = append(tmpEps, ep.endpointsv6...)
|
||||
|
||||
for _, endpoint := range tmpEps {
|
||||
if endpoint.Conditions.Serving == nil || !*endpoint.Conditions.Serving {
|
||||
continue
|
||||
}
|
||||
for _, address := range endpoint.Addresses {
|
||||
// 1. Compare the Nodename
|
||||
if endpoint.NodeName != nil && id == *endpoint.NodeName {
|
||||
if endpoint.Hostname != nil {
|
||||
log.Debug("found endpoint", "provider", ep.label, "ip", address, "hostname", *endpoint.Hostname, "nodename", *endpoint.NodeName)
|
||||
} else {
|
||||
log.Debug("found endpoint", "provider", ep.label, "ip", address, "nodename", *endpoint.NodeName)
|
||||
}
|
||||
localEndpoints = append(localEndpoints, address)
|
||||
for _, eps := range ep.slices {
|
||||
for _, endpoint := range eps.Endpoints {
|
||||
if !isServing(endpoint.Conditions) {
|
||||
continue
|
||||
}
|
||||
for _, address := range endpoint.Addresses {
|
||||
// 1. Compare the Nodename
|
||||
if endpoint.NodeName != nil && id == *endpoint.NodeName {
|
||||
if endpoint.Hostname != nil {
|
||||
log.Debug("found endpoint", "provider", ep.label, "ip", address, "hostname", *endpoint.Hostname, "nodename", *endpoint.NodeName)
|
||||
} else {
|
||||
log.Debug("found endpoint", "provider", ep.label, "ip", address, "nodename", *endpoint.NodeName)
|
||||
}
|
||||
localEndpoints = append(localEndpoints, address)
|
||||
continue
|
||||
}
|
||||
|
||||
// 2. Compare the Hostname (only useful if endpoint.NodeName is not available)
|
||||
if endpoint.Hostname != nil && id == *endpoint.Hostname {
|
||||
log.Debug("found endpoint", "provider", ep.label, "ip", address, "hostname", *endpoint.Hostname)
|
||||
localEndpoints = append(localEndpoints, address)
|
||||
// 2. Compare the Hostname (only useful if endpoint.NodeName is not available)
|
||||
if endpoint.NodeName == nil && endpoint.Hostname != nil && id == *endpoint.Hostname {
|
||||
log.Debug("found endpoint", "provider", ep.label, "ip", address, "hostname", *endpoint.Hostname)
|
||||
localEndpoints = append(localEndpoints, address)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -153,9 +167,11 @@ func (ep *Endpointslices) GetLabel() string {
|
||||
|
||||
func (ep *Endpointslices) ResolvePort(servicePort v1.ServicePort) int32 {
|
||||
return ResolvePortWithLookup(servicePort, func(name string) int32 {
|
||||
for _, p := range ep.ports {
|
||||
if p.Name != nil && *p.Name == name && p.Port != nil {
|
||||
return *p.Port
|
||||
for _, eps := range ep.slices {
|
||||
for _, p := range eps.Ports {
|
||||
if p.Name != nil && *p.Name == name && p.Port != nil {
|
||||
return *p.Port
|
||||
}
|
||||
}
|
||||
}
|
||||
return 0
|
||||
|
||||
149
pkg/endpoints/providers/endpointslices_test.go
Normal file
149
pkg/endpoints/providers/endpointslices_test.go
Normal file
@@ -0,0 +1,149 @@
|
||||
package providers
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
discoveryv1 "k8s.io/api/discovery/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
)
|
||||
|
||||
func TestEndpointslicesTracksAndDeletesSlices(t *testing.T) {
|
||||
provider := NewEndpointslices().(*Endpointslices)
|
||||
serving := true
|
||||
nodeName := "node-1"
|
||||
|
||||
slice1 := &discoveryv1.EndpointSlice{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "slice-1"},
|
||||
AddressType: discoveryv1.AddressTypeIPv4,
|
||||
Endpoints: []discoveryv1.Endpoint{{
|
||||
Addresses: []string{"10.0.0.1"},
|
||||
Conditions: discoveryv1.EndpointConditions{Serving: &serving},
|
||||
NodeName: &nodeName,
|
||||
}},
|
||||
}
|
||||
slice2 := &discoveryv1.EndpointSlice{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "slice-2"},
|
||||
AddressType: discoveryv1.AddressTypeIPv4,
|
||||
Endpoints: []discoveryv1.Endpoint{{
|
||||
Addresses: []string{"10.0.0.2"},
|
||||
Conditions: discoveryv1.EndpointConditions{Serving: &serving},
|
||||
NodeName: &nodeName,
|
||||
}},
|
||||
}
|
||||
|
||||
for _, slice := range []*discoveryv1.EndpointSlice{slice1, slice2} {
|
||||
if err := provider.LoadObject(slice, func() {}); err != nil {
|
||||
t.Fatalf("LoadObject returned error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
assertEndpoints(t, provider, []string{"10.0.0.1", "10.0.0.2"})
|
||||
assertLocalEndpoints(t, provider, nodeName, []string{"10.0.0.1", "10.0.0.2"})
|
||||
|
||||
if err := provider.DeleteObject(slice1); err != nil {
|
||||
t.Fatalf("DeleteObject returned error: %v", err)
|
||||
}
|
||||
assertEndpoints(t, provider, []string{"10.0.0.2"})
|
||||
assertLocalEndpoints(t, provider, nodeName, []string{"10.0.0.2"})
|
||||
|
||||
if err := provider.DeleteObject(slice2); err != nil {
|
||||
t.Fatalf("DeleteObject returned error: %v", err)
|
||||
}
|
||||
assertEndpoints(t, provider, nil)
|
||||
assertLocalEndpoints(t, provider, nodeName, nil)
|
||||
}
|
||||
|
||||
func TestEndpointslicesReplacingSliceUpdatesState(t *testing.T) {
|
||||
provider := NewEndpointslices().(*Endpointslices)
|
||||
first := &discoveryv1.EndpointSlice{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "slice-1"},
|
||||
AddressType: discoveryv1.AddressTypeIPv4,
|
||||
Endpoints: []discoveryv1.Endpoint{{Addresses: []string{"10.0.0.1"}}},
|
||||
}
|
||||
replacement := first.DeepCopy()
|
||||
replacement.Endpoints[0].Addresses = []string{"10.0.0.2"}
|
||||
|
||||
if err := provider.LoadObject(first, context.CancelFunc(func() {})); err != nil {
|
||||
t.Fatalf("LoadObject returned error: %v", err)
|
||||
}
|
||||
if err := provider.LoadObject(replacement, context.CancelFunc(func() {})); err != nil {
|
||||
t.Fatalf("LoadObject returned error: %v", err)
|
||||
}
|
||||
|
||||
assertEndpoints(t, provider, []string{"10.0.0.2"})
|
||||
}
|
||||
|
||||
func TestEndpointslicesEndpointConditions(t *testing.T) {
|
||||
yes, no := true, false
|
||||
nodeName := "node-1"
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
conditions discoveryv1.EndpointConditions
|
||||
want []string
|
||||
}{
|
||||
{"serving true", discoveryv1.EndpointConditions{Serving: &yes}, []string{"10.0.0.1"}},
|
||||
{"serving false", discoveryv1.EndpointConditions{Serving: &no}, nil},
|
||||
{"serving false overrides ready true", discoveryv1.EndpointConditions{Serving: &no, Ready: &yes}, nil},
|
||||
{"nil serving defers to ready true", discoveryv1.EndpointConditions{Ready: &yes}, []string{"10.0.0.1"}},
|
||||
{"nil serving defers to ready false", discoveryv1.EndpointConditions{Ready: &no}, nil},
|
||||
{"both nil is treated as ready", discoveryv1.EndpointConditions{}, []string{"10.0.0.1"}},
|
||||
}
|
||||
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
provider := NewEndpointslices().(*Endpointslices)
|
||||
slice := &discoveryv1.EndpointSlice{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "slice-1"},
|
||||
AddressType: discoveryv1.AddressTypeIPv4,
|
||||
Endpoints: []discoveryv1.Endpoint{{
|
||||
Addresses: []string{"10.0.0.1"},
|
||||
Conditions: test.conditions,
|
||||
NodeName: &nodeName,
|
||||
}},
|
||||
}
|
||||
if err := provider.LoadObject(slice, func() {}); err != nil {
|
||||
t.Fatalf("LoadObject returned error: %v", err)
|
||||
}
|
||||
// Cluster and Local policy have to agree on which endpoints are usable.
|
||||
assertEndpoints(t, provider, test.want)
|
||||
assertLocalEndpoints(t, provider, nodeName, test.want)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func assertEndpoints(t *testing.T, provider *Endpointslices, want []string) {
|
||||
t.Helper()
|
||||
got, err := provider.GetAllEndpoints()
|
||||
if err != nil {
|
||||
t.Fatalf("GetAllEndpoints returned error: %v", err)
|
||||
}
|
||||
assertStringSet(t, got, want)
|
||||
}
|
||||
|
||||
func assertLocalEndpoints(t *testing.T, provider *Endpointslices, nodeName string, want []string) {
|
||||
t.Helper()
|
||||
got, err := provider.GetLocalEndpoints(nodeName, &kubevip.Config{})
|
||||
if err != nil {
|
||||
t.Fatalf("GetLocalEndpoints returned error: %v", err)
|
||||
}
|
||||
assertStringSet(t, got, want)
|
||||
}
|
||||
|
||||
func assertStringSet(t *testing.T, got, want []string) {
|
||||
t.Helper()
|
||||
counts := map[string]int{}
|
||||
for _, value := range got {
|
||||
counts[value]++
|
||||
}
|
||||
for _, value := range want {
|
||||
counts[value]--
|
||||
}
|
||||
for value, count := range counts {
|
||||
if count != 0 {
|
||||
t.Fatalf("endpoint set mismatch for %q: got %v, want %v", value, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -18,6 +18,7 @@ type Provider interface {
|
||||
GetLabel() string
|
||||
UpdateServiceAnnotation(context.Context, string, string, *v1.Service, *kubernetes.Clientset) error
|
||||
LoadObject(runtime.Object, context.CancelFunc) error
|
||||
DeleteObject(runtime.Object) error
|
||||
// ResolvePort resolves a service port to the actual target port.
|
||||
// For named ports, it looks up the port number from the endpoint.
|
||||
// For numeric ports, it returns the port as-is.
|
||||
|
||||
319
pkg/endpoints/providers/providers_test.go
Normal file
319
pkg/endpoints/providers/providers_test.go
Normal file
@@ -0,0 +1,319 @@
|
||||
package providers
|
||||
|
||||
import (
|
||||
"net"
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
discoveryv1 "k8s.io/api/discovery/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/util/intstr"
|
||||
"k8s.io/client-go/kubernetes/fake"
|
||||
)
|
||||
|
||||
func TestEndpointProvidersParityForLocalAndAllEndpoints(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
nodeA := "node-a"
|
||||
nodeB := "node-b"
|
||||
serving := true
|
||||
v4Addresses := []discoveryv1.Endpoint{
|
||||
{Addresses: []string{"10.0.0.1"}, NodeName: &nodeA, Conditions: discoveryv1.EndpointConditions{Serving: &serving}},
|
||||
{Addresses: []string{"10.0.0.2"}, NodeName: &nodeB, Conditions: discoveryv1.EndpointConditions{Serving: &serving}},
|
||||
}
|
||||
v6Addresses := []discoveryv1.Endpoint{
|
||||
{Addresses: []string{"2001:db8::1"}, NodeName: &nodeA, Conditions: discoveryv1.EndpointConditions{Serving: &serving}},
|
||||
{Addresses: []string{"2001:db8::2"}, NodeName: &nodeB, Conditions: discoveryv1.EndpointConditions{Serving: &serving}},
|
||||
}
|
||||
|
||||
legacy := NewEndpoints()
|
||||
//nolint:staticcheck // this test covers the deprecated legacy Endpoints provider on purpose
|
||||
if err := legacy.LoadObject(&v1.Endpoints{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "service", Namespace: "default"},
|
||||
//nolint:staticcheck // deprecated legacy Endpoints API is the subject under test
|
||||
Subsets: []v1.EndpointSubset{{
|
||||
Addresses: []v1.EndpointAddress{
|
||||
{IP: "10.0.0.1", NodeName: &nodeA},
|
||||
{IP: "10.0.0.2", NodeName: &nodeB},
|
||||
{IP: "2001:db8::1", NodeName: &nodeA},
|
||||
{IP: "2001:db8::2", NodeName: &nodeB},
|
||||
},
|
||||
}},
|
||||
}, func() {}); err != nil {
|
||||
t.Fatalf("loading legacy Endpoints: %v", err)
|
||||
}
|
||||
|
||||
slices := NewEndpointslices()
|
||||
if err := slices.LoadObject(&discoveryv1.EndpointSlice{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "service-v4", Namespace: "default"},
|
||||
AddressType: discoveryv1.AddressTypeIPv4,
|
||||
Endpoints: v4Addresses,
|
||||
}, func() {}); err != nil {
|
||||
t.Fatalf("loading IPv4 EndpointSlice: %v", err)
|
||||
}
|
||||
if err := slices.LoadObject(&discoveryv1.EndpointSlice{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "service-v6", Namespace: "default"},
|
||||
AddressType: discoveryv1.AddressTypeIPv6,
|
||||
Endpoints: v6Addresses,
|
||||
}, func() {}); err != nil {
|
||||
t.Fatalf("loading IPv6 EndpointSlice: %v", err)
|
||||
}
|
||||
|
||||
legacyAll, err := legacy.GetAllEndpoints()
|
||||
if err != nil {
|
||||
t.Fatalf("legacy GetAllEndpoints() error = %v", err)
|
||||
}
|
||||
sliceAll, err := slices.GetAllEndpoints()
|
||||
if err != nil {
|
||||
t.Fatalf("EndpointSlice GetAllEndpoints() error = %v", err)
|
||||
}
|
||||
wantAll := endpointSet([]string{"10.0.0.1", "10.0.0.2", "2001:db8::1", "2001:db8::2"})
|
||||
if got := endpointSet(legacyAll); !reflect.DeepEqual(got, wantAll) {
|
||||
t.Errorf("legacy all endpoints = %v, want %v", got, wantAll)
|
||||
}
|
||||
if got := endpointSet(sliceAll); !reflect.DeepEqual(got, wantAll) {
|
||||
t.Errorf("EndpointSlice all endpoints = %v, want %v", got, wantAll)
|
||||
}
|
||||
|
||||
legacyLocal, err := legacy.GetLocalEndpoints(nodeA, &kubevip.Config{})
|
||||
if err != nil {
|
||||
t.Fatalf("legacy GetLocalEndpoints() error = %v", err)
|
||||
}
|
||||
sliceLocal, err := slices.GetLocalEndpoints(nodeA, &kubevip.Config{})
|
||||
if err != nil {
|
||||
t.Fatalf("EndpointSlice GetLocalEndpoints() error = %v", err)
|
||||
}
|
||||
wantLocal := endpointSet([]string{"10.0.0.1", "2001:db8::1"})
|
||||
if got := endpointSet(legacyLocal); !reflect.DeepEqual(got, wantLocal) {
|
||||
t.Errorf("legacy local endpoints = %v, want %v", got, wantLocal)
|
||||
}
|
||||
if got := endpointSet(sliceLocal); !reflect.DeepEqual(got, wantLocal) {
|
||||
t.Errorf("EndpointSlice local endpoints = %v, want %v", got, wantLocal)
|
||||
}
|
||||
|
||||
assertEndpointFamilies(t, legacyAll, 2, 2)
|
||||
assertEndpointFamilies(t, sliceAll, 2, 2)
|
||||
}
|
||||
|
||||
func TestEndpointSlicesLocalFilteringRequiresServingEndpoint(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
node := "node-a"
|
||||
serving := true
|
||||
notServing := false
|
||||
provider := NewEndpointslices()
|
||||
if err := provider.LoadObject(&discoveryv1.EndpointSlice{
|
||||
AddressType: discoveryv1.AddressTypeIPv4,
|
||||
Endpoints: []discoveryv1.Endpoint{
|
||||
{Addresses: []string{"10.0.0.1"}, NodeName: &node, Conditions: discoveryv1.EndpointConditions{Serving: &serving}},
|
||||
{Addresses: []string{"10.0.0.2"}, NodeName: &node, Conditions: discoveryv1.EndpointConditions{Serving: ¬Serving}},
|
||||
},
|
||||
}, func() {}); err != nil {
|
||||
t.Fatalf("loading EndpointSlice: %v", err)
|
||||
}
|
||||
|
||||
local, err := provider.GetLocalEndpoints(node, &kubevip.Config{})
|
||||
if err != nil {
|
||||
t.Fatalf("GetLocalEndpoints() error = %v", err)
|
||||
}
|
||||
if got, want := endpointSet(local), endpointSet([]string{"10.0.0.1"}); !reflect.DeepEqual(got, want) {
|
||||
t.Errorf("local endpoints = %v, want serving endpoints %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolvePortFromFakeClientObjects(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
//nolint:staticcheck // deprecated legacy Endpoints API is the subject under test
|
||||
legacyObject := &v1.Endpoints{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "service", Namespace: "default"},
|
||||
//nolint:staticcheck // deprecated legacy Endpoints API is the subject under test
|
||||
Subsets: []v1.EndpointSubset{{
|
||||
Ports: []v1.EndpointPort{{Name: "web", Port: 8080}},
|
||||
}},
|
||||
}
|
||||
legacyClient := fake.NewSimpleClientset(legacyObject)
|
||||
legacyLoaded, err := legacyClient.CoreV1().Endpoints("default").Get(t.Context(), "service", metav1.GetOptions{})
|
||||
if err != nil {
|
||||
t.Fatalf("getting fake legacy Endpoints: %v", err)
|
||||
}
|
||||
legacy := NewEndpoints()
|
||||
if err := legacy.LoadObject(legacyLoaded, func() {}); err != nil {
|
||||
t.Fatalf("loading fake legacy Endpoints: %v", err)
|
||||
}
|
||||
|
||||
portName := "web"
|
||||
port := int32(8081)
|
||||
sliceObject := &discoveryv1.EndpointSlice{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "service-slice", Namespace: "default"},
|
||||
AddressType: discoveryv1.AddressTypeIPv4,
|
||||
Ports: []discoveryv1.EndpointPort{{Name: &portName, Port: &port}},
|
||||
}
|
||||
sliceClient := fake.NewSimpleClientset(sliceObject)
|
||||
sliceLoaded, err := sliceClient.DiscoveryV1().EndpointSlices("default").Get(t.Context(), "service-slice", metav1.GetOptions{})
|
||||
if err != nil {
|
||||
t.Fatalf("getting fake EndpointSlice: %v", err)
|
||||
}
|
||||
slices := NewEndpointslices()
|
||||
if err := slices.LoadObject(sliceLoaded, func() {}); err != nil {
|
||||
t.Fatalf("loading fake EndpointSlice: %v", err)
|
||||
}
|
||||
|
||||
namedPort := v1.ServicePort{Port: 80, TargetPort: intstr.FromString("web")}
|
||||
if got := legacy.ResolvePort(namedPort); got != 8080 {
|
||||
t.Errorf("legacy ResolvePort() = %d, want 8080", got)
|
||||
}
|
||||
if got := slices.ResolvePort(namedPort); got != 8081 {
|
||||
t.Errorf("EndpointSlice ResolvePort() = %d, want 8081", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolvePortWithLookup(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
port v1.ServicePort
|
||||
lookup func(string) int32
|
||||
want int32
|
||||
}{
|
||||
{
|
||||
name: "numeric target port wins",
|
||||
port: v1.ServicePort{Port: 80, TargetPort: intstr.FromInt(8080)},
|
||||
lookup: func(string) int32 { return 9090 },
|
||||
want: 8080,
|
||||
},
|
||||
{
|
||||
name: "named target port is looked up",
|
||||
port: v1.ServicePort{Port: 80, TargetPort: intstr.FromString("web")},
|
||||
lookup: func(name string) int32 {
|
||||
if name == "web" {
|
||||
return 8081
|
||||
}
|
||||
return 0
|
||||
},
|
||||
want: 8081,
|
||||
},
|
||||
{
|
||||
name: "missing named target falls back to service port",
|
||||
port: v1.ServicePort{Port: 80, TargetPort: intstr.FromString("missing")},
|
||||
lookup: func(string) int32 { return 0 },
|
||||
want: 80,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if got := ResolvePortWithLookup(tt.port, tt.lookup); got != tt.want {
|
||||
t.Errorf("ResolvePortWithLookup() = %d, want %d", got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func endpointSet(endpoints []string) map[string]struct{} {
|
||||
result := make(map[string]struct{}, len(endpoints))
|
||||
for _, endpoint := range endpoints {
|
||||
result[endpoint] = struct{}{}
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
func assertEndpointFamilies(t *testing.T, endpoints []string, wantIPv4, wantIPv6 int) {
|
||||
t.Helper()
|
||||
ipv4, ipv6 := 0, 0
|
||||
for _, endpoint := range endpoints {
|
||||
ip := net.ParseIP(endpoint)
|
||||
if ip == nil {
|
||||
t.Errorf("endpoint %q is not an IP address", endpoint)
|
||||
continue
|
||||
}
|
||||
if ip.To4() != nil {
|
||||
ipv4++
|
||||
} else {
|
||||
ipv6++
|
||||
}
|
||||
}
|
||||
if ipv4 != wantIPv4 || ipv6 != wantIPv6 {
|
||||
t.Errorf("endpoint families = IPv4 %d, IPv6 %d; want IPv4 %d, IPv6 %d", ipv4, ipv6, wantIPv4, wantIPv6)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEndpointProvidersPreferNodeNameOverHostname(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
nodeA := "node-a"
|
||||
nodeB := "node-b"
|
||||
serving := true
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
load func(Provider) error
|
||||
}{
|
||||
{
|
||||
name: "legacy Endpoints",
|
||||
load: func(provider Provider) error {
|
||||
//nolint:staticcheck // the legacy provider is deliberately under test
|
||||
return provider.LoadObject(&v1.Endpoints{
|
||||
Subsets: []v1.EndpointSubset{{
|
||||
Addresses: []v1.EndpointAddress{{
|
||||
IP: "10.0.0.1",
|
||||
NodeName: &nodeB,
|
||||
Hostname: nodeA,
|
||||
}},
|
||||
}},
|
||||
}, func() {})
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "EndpointSlice",
|
||||
load: func(provider Provider) error {
|
||||
hostname := nodeA
|
||||
return provider.LoadObject(&discoveryv1.EndpointSlice{
|
||||
AddressType: discoveryv1.AddressTypeIPv4,
|
||||
Endpoints: []discoveryv1.Endpoint{{
|
||||
Addresses: []string{"10.0.0.1"},
|
||||
NodeName: &nodeB,
|
||||
Hostname: &hostname,
|
||||
Conditions: discoveryv1.EndpointConditions{Serving: &serving},
|
||||
}},
|
||||
}, func() {})
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
tt := tt
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
var provider Provider
|
||||
if tt.name == "legacy Endpoints" {
|
||||
provider = NewEndpoints()
|
||||
} else {
|
||||
provider = NewEndpointslices()
|
||||
}
|
||||
if err := tt.load(provider); err != nil {
|
||||
t.Fatalf("LoadObject() error = %v", err)
|
||||
}
|
||||
|
||||
local, err := provider.GetLocalEndpoints(nodeA, &kubevip.Config{})
|
||||
if err != nil {
|
||||
t.Fatalf("GetLocalEndpoints(%q) error = %v", nodeA, err)
|
||||
}
|
||||
if len(local) != 0 {
|
||||
t.Fatalf("GetLocalEndpoints(%q) = %v, want no endpoints", nodeA, local)
|
||||
}
|
||||
|
||||
local, err = provider.GetLocalEndpoints(nodeB, &kubevip.Config{})
|
||||
if err != nil {
|
||||
t.Fatalf("GetLocalEndpoints(%q) error = %v", nodeB, err)
|
||||
}
|
||||
if got, want := endpointSet(local), endpointSet([]string{"10.0.0.1"}); !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("GetLocalEndpoints(%q) = %v, want %v", nodeB, got, want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -15,7 +15,6 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/pkg/errors"
|
||||
"github.com/sirupsen/logrus"
|
||||
)
|
||||
|
||||
const (
|
||||
@@ -28,7 +27,6 @@ const (
|
||||
)
|
||||
|
||||
func TestMain(m *testing.M) {
|
||||
logrus.SetLevel(logrus.DebugLevel)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
expectSuccess(startEtcd(ctx), "starting etcd")
|
||||
|
||||
@@ -2,17 +2,20 @@ package instance
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
|
||||
log "log/slog"
|
||||
|
||||
"github.com/vishvananda/netlink"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/arp"
|
||||
"github.com/kube-vip/kube-vip/pkg/cluster"
|
||||
@@ -46,18 +49,21 @@ type Instance struct {
|
||||
DHCPv6Client vip.DHCPClient
|
||||
macvlanName string
|
||||
dhcpBroadcast bool
|
||||
dhcpInterfaceOwned atomic.Bool
|
||||
|
||||
// Service use Vlan
|
||||
IsVLAN bool
|
||||
VLANInterface string
|
||||
vlanOwned atomic.Bool
|
||||
|
||||
// External Gateway IP the service is forwarded from
|
||||
UPNPGatewayIPs []string
|
||||
|
||||
// Kubernetes service mapping
|
||||
ServiceSnapshot *v1.Service
|
||||
|
||||
dnsAddresses []string
|
||||
ServiceUID types.UID
|
||||
ServiceAddresses []string
|
||||
ServiceSnapshot *v1.Service
|
||||
cleanupInfo *ServiceCleanupInfo
|
||||
|
||||
// AddCalled determined that ActionAdd was already performed for the instance
|
||||
AddCalled bool
|
||||
@@ -67,6 +73,48 @@ type Instance struct {
|
||||
LabelAdded bool
|
||||
}
|
||||
|
||||
func (instance *Instance) UID() types.UID {
|
||||
return instance.ServiceUID
|
||||
}
|
||||
|
||||
func (instance *Instance) Addresses() []string {
|
||||
if instance.ServiceAddresses != nil {
|
||||
return append([]string(nil), instance.ServiceAddresses...)
|
||||
}
|
||||
addresses, _ := FetchServiceAddresses(instance.ServiceSnapshot)
|
||||
return addresses
|
||||
}
|
||||
|
||||
type ServiceCleanupInfo struct {
|
||||
Namespace string
|
||||
Name string
|
||||
Lease string
|
||||
ExternalTrafficPolicy v1.ServiceExternalTrafficPolicy
|
||||
}
|
||||
|
||||
// CleanupInfo returns Service policy captured when the instance was created.
|
||||
func (instance *Instance) CleanupInfo() (ServiceCleanupInfo, bool) {
|
||||
if instance == nil {
|
||||
return ServiceCleanupInfo{}, false
|
||||
}
|
||||
if instance.cleanupInfo != nil {
|
||||
return *instance.cleanupInfo, true
|
||||
}
|
||||
if instance.ServiceSnapshot == nil {
|
||||
return ServiceCleanupInfo{}, false
|
||||
}
|
||||
return serviceCleanupInfo(instance.ServiceSnapshot), true
|
||||
}
|
||||
|
||||
func serviceCleanupInfo(service *v1.Service) ServiceCleanupInfo {
|
||||
return ServiceCleanupInfo{
|
||||
Namespace: service.Namespace,
|
||||
Name: service.Name,
|
||||
Lease: service.Annotations[kubevip.ServiceLease],
|
||||
ExternalTrafficPolicy: service.Spec.ExternalTrafficPolicy,
|
||||
}
|
||||
}
|
||||
|
||||
type Port struct {
|
||||
Port uint16
|
||||
Type string
|
||||
@@ -78,16 +126,25 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
instanceAddresses, instanceHostnames := FetchServiceAddresses(svc)
|
||||
log.Info("new instance", "namespace", svc.Namespace, "service", svc.Name, "addresses", instanceAddresses, "hostnames", instanceHostnames)
|
||||
|
||||
cleanupInfo := serviceCleanupInfo(svc)
|
||||
instance := &Instance{
|
||||
ServiceUID: svc.UID,
|
||||
ServiceAddresses: append([]string(nil), instanceAddresses...),
|
||||
ServiceSnapshot: svc,
|
||||
cleanupInfo: &cleanupInfo,
|
||||
}
|
||||
if err := instance.initialize(ctx, svc, config, intfMgr, arpMgr, routeMgr, nodeLabelMgr, wg, instanceAddresses, instanceHostnames); err != nil {
|
||||
return nil, errors.Join(err, instance.CleanupLinkAttachments())
|
||||
}
|
||||
return instance, nil
|
||||
}
|
||||
|
||||
func (instance *Instance) initialize(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
intfMgr *networkinterface.Manager, arpMgr *arp.Manager, routeMgr *route.Manager,
|
||||
nodeLabelMgr node.Labeler, wg *sync.WaitGroup, instanceAddresses, instanceHostnames []string) error {
|
||||
var newVips []*kubevip.Config
|
||||
var link netlink.Link
|
||||
var err error
|
||||
var dnsAddresses []string
|
||||
|
||||
// Create new service
|
||||
instance := &Instance{
|
||||
ServiceSnapshot: svc,
|
||||
dnsAddresses: dnsAddresses,
|
||||
}
|
||||
|
||||
for _, address := range instanceAddresses {
|
||||
// Detect if we're using a specific interface for services
|
||||
@@ -143,10 +200,10 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
|
||||
if link == nil {
|
||||
if link, err = netlink.LinkByName(svcInterface); err != nil {
|
||||
return nil, fmt.Errorf("failed to get interface %s: %w", svcInterface, err)
|
||||
return fmt.Errorf("failed to get interface %s: %w", svcInterface, err)
|
||||
}
|
||||
if link == nil {
|
||||
return nil, fmt.Errorf("failed to get interface %s", svcInterface)
|
||||
return fmt.Errorf("failed to get interface %s", svcInterface)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -163,7 +220,7 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
}
|
||||
|
||||
if (config.Address != "" || config.VIP != "") && (ipv4AutoSubnet || ipv6AutoSubnet) {
|
||||
return nil, fmt.Errorf("auto subnet discovery cannot be used if VIP address was provided")
|
||||
return fmt.Errorf("auto subnet discovery cannot be used if VIP address was provided")
|
||||
}
|
||||
|
||||
subnet := ""
|
||||
@@ -172,7 +229,7 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
if ipv4AutoSubnet {
|
||||
subnet, err = autoFindSubnet(link, address)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to automatically find subnet for service %s/%s with IP address %s on interface %s: %w", svc.Namespace, svc.Name, address, svcInterface, err)
|
||||
return fmt.Errorf("failed to automatically find subnet for service %s/%s with IP address %s on interface %s: %w", svc.Namespace, svc.Name, address, svcInterface, err)
|
||||
}
|
||||
} else {
|
||||
if cidrs[0] != "" && cidrs[0] != kubevip.Auto {
|
||||
@@ -185,7 +242,7 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
if ipv6AutoSubnet {
|
||||
subnet, err = autoFindSubnet(link, address)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to automatically find subnet for service %s/%s with IP address %s on interface %s: %w", svc.Namespace, svc.Name, address, svcInterface, err)
|
||||
return fmt.Errorf("failed to automatically find subnet for service %s/%s with IP address %s on interface %s: %w", svc.Namespace, svc.Name, address, svcInterface, err)
|
||||
}
|
||||
} else {
|
||||
if len(cidrs) > 1 && cidrs[1] != "" && cidrs[1] != kubevip.Auto {
|
||||
@@ -198,23 +255,26 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
|
||||
// Generate new Virtual IP configuration
|
||||
newVips = append(newVips, &kubevip.Config{
|
||||
VIP: address,
|
||||
Interface: svcInterface,
|
||||
SingleNode: true,
|
||||
EnableARP: config.EnableARP,
|
||||
EnableBGP: config.EnableBGP,
|
||||
VIPSubnet: subnet,
|
||||
EnableRoutingTable: config.EnableRoutingTable,
|
||||
RoutingTableID: config.RoutingTableID,
|
||||
RoutingTableType: config.RoutingTableType,
|
||||
RoutingProtocol: config.RoutingProtocol,
|
||||
ArpBroadcastRate: config.ArpBroadcastRate,
|
||||
EnableServiceSecurity: config.EnableServiceSecurity,
|
||||
DNSMode: config.DNSMode,
|
||||
DHCPMode: config.DHCPMode,
|
||||
DHCPBackoffAttempts: config.DHCPBackoffAttempts,
|
||||
DisableServiceUpdates: config.DisableServiceUpdates,
|
||||
EnableServicesElection: config.EnableServicesElection,
|
||||
VIP: address,
|
||||
Interface: svcInterface,
|
||||
SingleNode: true,
|
||||
EnableARP: config.EnableARP,
|
||||
EnableBGP: config.EnableBGP,
|
||||
BGPAttachIPToInterface: config.BGPAttachIPToInterface,
|
||||
VIPSubnet: subnet,
|
||||
EnableRoutingTable: config.EnableRoutingTable,
|
||||
RoutingTableID: config.RoutingTableID,
|
||||
RoutingTableType: config.RoutingTableType,
|
||||
RoutingProtocol: config.RoutingProtocol,
|
||||
SkipDAD: config.SkipDAD,
|
||||
ArpBroadcastRate: config.ArpBroadcastRate,
|
||||
EnableServiceSecurity: config.EnableServiceSecurity,
|
||||
DNSMode: config.DNSMode,
|
||||
DHCPMode: config.DHCPMode,
|
||||
DHCPBackoffAttempts: config.DHCPBackoffAttempts,
|
||||
DisableServiceUpdates: config.DisableServiceUpdates,
|
||||
EnableServicesElection: config.EnableServicesElection,
|
||||
// cleanupVIPs reads this from the per-VIP config, so Service VIPs need it too.
|
||||
PreserveVIPOnLeadershipLoss: config.PreserveVIPOnLeadershipLoss,
|
||||
KubernetesLeaderElection: kubevip.KubernetesLeaderElection{
|
||||
EnableLeaderElection: config.EnableLeaderElection,
|
||||
@@ -254,10 +314,10 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
|
||||
if link == nil {
|
||||
if link, err = netlink.LinkByName(svcInterface); err != nil {
|
||||
return nil, fmt.Errorf("failed to get interface %s: %w", svcInterface, err)
|
||||
return fmt.Errorf("failed to get interface %s: %w", svcInterface, err)
|
||||
}
|
||||
if link == nil {
|
||||
return nil, fmt.Errorf("failed to get interface %s", svcInterface)
|
||||
return fmt.Errorf("failed to get interface %s", svcInterface)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -268,11 +328,13 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
SingleNode: true,
|
||||
EnableARP: config.EnableARP,
|
||||
EnableBGP: config.EnableBGP,
|
||||
BGPAttachIPToInterface: config.BGPAttachIPToInterface,
|
||||
VIPSubnet: config.VIPSubnet,
|
||||
EnableRoutingTable: config.EnableRoutingTable,
|
||||
RoutingTableID: config.RoutingTableID,
|
||||
RoutingTableType: config.RoutingTableType,
|
||||
RoutingProtocol: config.RoutingProtocol,
|
||||
SkipDAD: config.SkipDAD,
|
||||
ArpBroadcastRate: config.ArpBroadcastRate,
|
||||
EnableServiceSecurity: config.EnableServiceSecurity,
|
||||
DNSMode: config.DNSMode,
|
||||
@@ -291,7 +353,7 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
if requestedIP != "" {
|
||||
requestedIPs := strings.Split(requestedIP, ",")
|
||||
if len(requestedIPs) > 2 {
|
||||
return nil, fmt.Errorf("annotation %q cannot request more than one IPv4 and one Ipv6 address", kubevip.RequestedIP)
|
||||
return fmt.Errorf("annotation %q cannot request more than one IPv4 and one Ipv6 address", kubevip.RequestedIP)
|
||||
}
|
||||
for _, ip := range requestedIPs {
|
||||
netip := net.ParseIP(ip)
|
||||
@@ -330,36 +392,40 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
// If this was purposely created with the address '0.0.0.0', or '::'
|
||||
// we will create a macvlan on the main interface and a DHCP client
|
||||
if len(instanceAddresses) > 2 && (slices.Contains(instanceAddresses, "0.0.0.0") || slices.Contains(instanceAddresses, "::")) {
|
||||
return nil, fmt.Errorf("DHCP cannot be used if more than 2 addresses (one IPv4 and one IPv6) were specified")
|
||||
return fmt.Errorf("DHCP cannot be used if more than 2 addresses (one IPv4 and one IPv6) were specified")
|
||||
}
|
||||
for i := range instance.VIPConfigs {
|
||||
if instance.VIPConfigs[i].VIP == "0.0.0.0" {
|
||||
err := instance.startDHCP(ctx, i, config.DHCPBackoffAttempts, wg)
|
||||
for index := range instance.VIPConfigs {
|
||||
if instance.VIPConfigs[index].VIP == "0.0.0.0" {
|
||||
err := instance.startDHCP(ctx, index, config.DHCPBackoffAttempts, wg)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
return err
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case err := <-instance.DHCPv4Client.ErrorChannel():
|
||||
return nil, fmt.Errorf("error starting DHCPv4 for %s/%s: error: %s",
|
||||
return fmt.Errorf("error starting DHCPv4 for %s/%s: error: %s",
|
||||
instance.ServiceSnapshot.Namespace, instance.ServiceSnapshot.Name, err)
|
||||
case ip := <-instance.DHCPv4Client.IPChannel():
|
||||
instance.VIPConfigs[i].Interface = instance.DHCPInterface
|
||||
instance.VIPConfigs[i].VIP = ip
|
||||
instance.VIPConfigs[index].Interface = instance.DHCPInterface
|
||||
instance.VIPConfigs[index].VIP = ip
|
||||
instance.DHCPInterfaceIPv4 = ip
|
||||
}
|
||||
}
|
||||
if instance.VIPConfigs[i].VIP == "::" {
|
||||
err := instance.startDHCP(ctx, i, config.DHCPBackoffAttempts, wg)
|
||||
if instance.VIPConfigs[index].VIP == "::" {
|
||||
err := instance.startDHCP(ctx, index, config.DHCPBackoffAttempts, wg)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
return err
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case err := <-instance.DHCPv6Client.ErrorChannel():
|
||||
return nil, fmt.Errorf("error starting DHCPv6 for %s/%s: error: %s",
|
||||
return fmt.Errorf("error starting DHCPv6 for %s/%s: error: %s",
|
||||
instance.ServiceSnapshot.Namespace, instance.ServiceSnapshot.Name, err)
|
||||
case ip := <-instance.DHCPv6Client.IPChannel():
|
||||
instance.VIPConfigs[i].Interface = instance.DHCPInterface
|
||||
instance.VIPConfigs[i].VIP = ip
|
||||
instance.VIPConfigs[index].Interface = instance.DHCPInterface
|
||||
instance.VIPConfigs[index].VIP = ip
|
||||
instance.DHCPInterfaceIPv6 = ip
|
||||
}
|
||||
}
|
||||
@@ -367,56 +433,56 @@ func NewInstance(ctx context.Context, svc *v1.Service, config *kubevip.Config,
|
||||
ddnsAnnotation, exists := svc.Annotations[kubevip.ServiceDDNS]
|
||||
|
||||
if exists {
|
||||
instance.VIPConfigs[i].DDNS, err = strconv.ParseBool(ddnsAnnotation)
|
||||
instance.VIPConfigs[index].DDNS, err = strconv.ParseBool(ddnsAnnotation)
|
||||
if err != nil {
|
||||
log.Error("Failed to add service", "err", err)
|
||||
return nil, err
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
if len(svc.Spec.IPFamilies) > 0 {
|
||||
if len(svc.Spec.IPFamilies) > 1 {
|
||||
instance.VIPConfigs[i].DHCPMode = utils.DualFamily
|
||||
instance.VIPConfigs[i].DNSMode = utils.DualFamily
|
||||
instance.VIPConfigs[index].DHCPMode = utils.DualFamily
|
||||
instance.VIPConfigs[index].DNSMode = utils.DualFamily
|
||||
switch *svc.Spec.IPFamilyPolicy {
|
||||
case v1.IPFamilyPolicyRequireDualStack:
|
||||
instance.VIPConfigs[i].IsDualStack = true
|
||||
instance.VIPConfigs[i].RequireDualStack = true
|
||||
instance.VIPConfigs[index].IsDualStack = true
|
||||
instance.VIPConfigs[index].RequireDualStack = true
|
||||
case v1.IPFamilyPolicyPreferDualStack:
|
||||
instance.VIPConfigs[i].IsDualStack = true
|
||||
instance.VIPConfigs[i].RequireDualStack = false
|
||||
instance.VIPConfigs[index].IsDualStack = true
|
||||
instance.VIPConfigs[index].RequireDualStack = false
|
||||
default:
|
||||
instance.VIPConfigs[i].IsDualStack = false
|
||||
instance.VIPConfigs[i].RequireDualStack = false
|
||||
instance.VIPConfigs[index].IsDualStack = false
|
||||
instance.VIPConfigs[index].RequireDualStack = false
|
||||
}
|
||||
} else {
|
||||
if strings.EqualFold(string(svc.Spec.IPFamilies[0]), utils.IPv4Family) {
|
||||
instance.VIPConfigs[i].DHCPMode = strings.ToLower(utils.IPv4Family)
|
||||
instance.VIPConfigs[i].DNSMode = strings.ToLower(utils.IPv4Family)
|
||||
instance.VIPConfigs[index].DHCPMode = strings.ToLower(utils.IPv4Family)
|
||||
instance.VIPConfigs[index].DNSMode = strings.ToLower(utils.IPv4Family)
|
||||
} else {
|
||||
instance.VIPConfigs[i].DHCPMode = strings.ToLower(utils.IPv6Family)
|
||||
instance.VIPConfigs[i].DNSMode = strings.ToLower(utils.IPv6Family)
|
||||
instance.VIPConfigs[index].DHCPMode = strings.ToLower(utils.IPv6Family)
|
||||
instance.VIPConfigs[index].DNSMode = strings.ToLower(utils.IPv6Family)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
instance.VIPConfigs[i].EgressWithNftables = config.EgressWithNftables
|
||||
instance.VIPConfigs[index].EgressWithNftables = config.EgressWithNftables
|
||||
|
||||
c, err := cluster.InitCluster(instance.VIPConfigs[i], false, intfMgr, arpMgr, routeMgr, nodeLabelMgr)
|
||||
c, err := cluster.InitCluster(instance.VIPConfigs[index], false, intfMgr, arpMgr, routeMgr, nodeLabelMgr)
|
||||
if err != nil {
|
||||
log.Error("failed to add service", "err", err)
|
||||
return nil, err
|
||||
return err
|
||||
}
|
||||
|
||||
for i := range c.Network {
|
||||
c.Network[i].SetServicePorts(svc)
|
||||
for networkIndex := range c.Network {
|
||||
c.Network[networkIndex].SetServicePorts(svc)
|
||||
}
|
||||
|
||||
instance.Clusters = append(instance.Clusters, c)
|
||||
log.Info("(svcs) adding VIP", "ip", instance.VIPConfigs[i].VIP, "interface", instance.VIPConfigs[i].Interface, "namespace", svc.Namespace, "name", svc.Name)
|
||||
log.Info("(svcs) adding VIP", "ip", instance.VIPConfigs[index].VIP, "interface", instance.VIPConfigs[index].Interface, "namespace", svc.Namespace, "name", svc.Name)
|
||||
}
|
||||
|
||||
return instance, nil
|
||||
return nil
|
||||
}
|
||||
|
||||
func autoFindInterface(ip string) (netlink.Link, error) {
|
||||
@@ -476,16 +542,17 @@ func getAutoInterfaceName(link netlink.Link, defaultInterface string) string {
|
||||
return link.Attrs().Name
|
||||
}
|
||||
|
||||
func (i *Instance) addVLAN(parentInterface string, tag int) error {
|
||||
var parent netlink.Link
|
||||
|
||||
func (instance *Instance) addVLAN(parentInterface string, tag int) error {
|
||||
interfaceName := fmt.Sprintf("%s.%d", parentInterface, tag)
|
||||
parent, err := netlink.LinkByName(parentInterface)
|
||||
if err != nil {
|
||||
return fmt.Errorf("finding VLAN parent interface %s: %w", parentInterface, err)
|
||||
}
|
||||
iface, err := netlink.LinkByName(interfaceName)
|
||||
if err != nil {
|
||||
// check if parent interface doesnt exist
|
||||
parent, err = netlink.LinkByName(parentInterface)
|
||||
if err != nil {
|
||||
return fmt.Errorf("error finding VLAN parent interface %s: %v", parentInterface, err)
|
||||
var notFound netlink.LinkNotFoundError
|
||||
if !errors.As(err, ¬Found) {
|
||||
return fmt.Errorf("finding VLAN interface %s: %w", interfaceName, err)
|
||||
}
|
||||
|
||||
log.Info("Creating new VLAN interface", "interface", interfaceName)
|
||||
@@ -503,6 +570,7 @@ func (i *Instance) addVLAN(parentInterface string, tag int) error {
|
||||
if err != nil {
|
||||
return fmt.Errorf("could not add VLAN %s: %v", interfaceName, err)
|
||||
}
|
||||
instance.vlanOwned.Store(true)
|
||||
|
||||
err = netlink.LinkSetUp(vlan)
|
||||
if err != nil {
|
||||
@@ -521,26 +589,96 @@ func (i *Instance) addVLAN(parentInterface string, tag int) error {
|
||||
}
|
||||
}
|
||||
|
||||
i.VLANInterface = interfaceName
|
||||
i.IsVLAN = true
|
||||
instance.VLANInterface = interfaceName
|
||||
instance.IsVLAN = true
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (i *Instance) startDHCP(ctx context.Context, index int, backoffAttempts uint, wg *sync.WaitGroup) error {
|
||||
if len(i.VIPConfigs) > 2 {
|
||||
return fmt.Errorf("DHCP can be used with 2 VIP config maximally, got: %v", len(i.VIPConfigs))
|
||||
// CleanupLinkAttachments stops this instance's DHCP clients and removes only
|
||||
// VLAN or macvlan links created by this instance that are not used by a
|
||||
// remaining Service instance.
|
||||
func (instance *Instance) CleanupLinkAttachments(remaining ...*Instance) error {
|
||||
var errs []error
|
||||
if instance.DHCPv4Client != nil {
|
||||
instance.DHCPv4Client.Stop()
|
||||
}
|
||||
parent, err := netlink.LinkByName(i.VIPConfigs[index].Interface)
|
||||
if instance.DHCPv6Client != nil {
|
||||
instance.DHCPv6Client.Stop()
|
||||
}
|
||||
if instance.dhcpInterfaceOwned.Load() {
|
||||
if transferLinkAttachmentOwnership(instance.DHCPInterface, remaining, false) {
|
||||
instance.dhcpInterfaceOwned.Store(false)
|
||||
} else if err := deleteOwnedLink(instance.DHCPInterface, "DHCP"); err != nil {
|
||||
errs = append(errs, err)
|
||||
} else {
|
||||
instance.dhcpInterfaceOwned.Store(false)
|
||||
}
|
||||
}
|
||||
if instance.vlanOwned.Load() {
|
||||
if transferLinkAttachmentOwnership(instance.VLANInterface, remaining, true) {
|
||||
instance.vlanOwned.Store(false)
|
||||
} else if err := deleteOwnedLink(instance.VLANInterface, "VLAN"); err != nil {
|
||||
errs = append(errs, err)
|
||||
} else {
|
||||
instance.vlanOwned.Store(false)
|
||||
}
|
||||
}
|
||||
return errors.Join(errs...)
|
||||
}
|
||||
|
||||
func transferLinkAttachmentOwnership(name string, instances []*Instance, vlan bool) bool {
|
||||
if name == "" {
|
||||
return false
|
||||
}
|
||||
for _, instance := range instances {
|
||||
if instance == nil {
|
||||
continue
|
||||
}
|
||||
if vlan && instance.IsVLAN && instance.VLANInterface == name {
|
||||
instance.vlanOwned.Store(true)
|
||||
return true
|
||||
}
|
||||
if !vlan && (instance.IsDHCPv4 || instance.IsDHCPv6) && instance.DHCPInterface == name {
|
||||
instance.dhcpInterfaceOwned.Store(true)
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func deleteOwnedLink(name, kind string) error {
|
||||
if name == "" {
|
||||
return nil
|
||||
}
|
||||
link, err := netlink.LinkByName(name)
|
||||
if err != nil {
|
||||
var notFound netlink.LinkNotFoundError
|
||||
if errors.As(err, ¬Found) {
|
||||
return nil
|
||||
}
|
||||
return fmt.Errorf("find %s interface %q: %w", kind, name, err)
|
||||
}
|
||||
if err := netlink.LinkDel(link); err != nil {
|
||||
return fmt.Errorf("delete %s interface %q: %w", kind, name, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (instance *Instance) startDHCP(ctx context.Context, index int, backoffAttempts uint, wg *sync.WaitGroup) error {
|
||||
if len(instance.VIPConfigs) > 2 {
|
||||
return fmt.Errorf("DHCP can be used with 2 VIP config maximally, got: %v", len(instance.VIPConfigs))
|
||||
}
|
||||
parent, err := netlink.LinkByName(instance.VIPConfigs[index].Interface)
|
||||
if err != nil {
|
||||
return fmt.Errorf("error finding VIP Interface, for building DHCP Link : %v", err)
|
||||
}
|
||||
|
||||
interfaceName := i.macvlanName
|
||||
interfaceName := instance.macvlanName
|
||||
|
||||
if interfaceName == "" {
|
||||
// Generate name from UID
|
||||
interfaceName = fmt.Sprintf("vip-%s", i.ServiceSnapshot.UID[0:8])
|
||||
interfaceName = fmt.Sprintf("vip-%s", instance.UID()[0:8])
|
||||
}
|
||||
|
||||
// Check if the interface doesn't exist first
|
||||
@@ -548,8 +686,8 @@ func (i *Instance) startDHCP(ctx context.Context, index int, backoffAttempts uin
|
||||
if err != nil {
|
||||
log.Info("creating new macvlan interface for DHCP", "interface", interfaceName)
|
||||
|
||||
hwaddr, err := net.ParseMAC(i.DHCPInterfaceHwaddr)
|
||||
if i.DHCPInterfaceHwaddr != "" && err != nil {
|
||||
hwaddr, err := net.ParseMAC(instance.DHCPInterfaceHwaddr)
|
||||
if instance.DHCPInterfaceHwaddr != "" && err != nil {
|
||||
return err
|
||||
} else if hwaddr == nil {
|
||||
hwaddr, err = net.ParseMAC(vip.GenerateMac())
|
||||
@@ -572,6 +710,7 @@ func (i *Instance) startDHCP(ctx context.Context, index int, backoffAttempts uin
|
||||
if err != nil {
|
||||
return fmt.Errorf("could not add %s: %v", interfaceName, err)
|
||||
}
|
||||
instance.dhcpInterfaceOwned.Store(true)
|
||||
|
||||
err = netlink.LinkSetUp(mac)
|
||||
if err != nil {
|
||||
@@ -587,7 +726,7 @@ func (i *Instance) startDHCP(ctx context.Context, index int, backoffAttempts uin
|
||||
}
|
||||
|
||||
var initRebootFlag bool
|
||||
ip := net.ParseIP(i.VIPConfigs[index].VIP)
|
||||
ip := net.ParseIP(instance.VIPConfigs[index].VIP)
|
||||
|
||||
var client vip.DHCPClient
|
||||
if ip.To4() != nil {
|
||||
@@ -595,14 +734,14 @@ func (i *Instance) startDHCP(ctx context.Context, index int, backoffAttempts uin
|
||||
rpfilterSetting := "0"
|
||||
|
||||
// Check if we need to set an override rp_filter value for the interface
|
||||
if i.ServiceSnapshot.Annotations[kubevip.RPFilter] != "" {
|
||||
if instance.ServiceSnapshot.Annotations[kubevip.RPFilter] != "" {
|
||||
// Check the rp_filter value
|
||||
rpFilter, err := strconv.Atoi(i.ServiceSnapshot.Annotations[kubevip.RPFilter])
|
||||
rpFilter, err := strconv.Atoi(instance.ServiceSnapshot.Annotations[kubevip.RPFilter])
|
||||
if err != nil {
|
||||
log.Error("[DHCP] unable to process rp_filter", "value", rpFilter)
|
||||
} else {
|
||||
if rpFilter >= 0 && rpFilter < 3 { // Ensure the value is 0,1,2
|
||||
rpfilterSetting = i.ServiceSnapshot.Annotations[kubevip.RPFilter]
|
||||
rpfilterSetting = instance.ServiceSnapshot.Annotations[kubevip.RPFilter]
|
||||
} else {
|
||||
log.Error("[DHCP] rp_filter value not within range 0-2", "value", rpFilter)
|
||||
}
|
||||
@@ -614,49 +753,50 @@ func (i *Instance) startDHCP(ctx context.Context, index int, backoffAttempts uin
|
||||
log.Error("[DHCP] unable to write rp_filter", "value", rpfilterSetting, "err", err)
|
||||
}
|
||||
|
||||
if i.DHCPInterfaceIPv4 != "" {
|
||||
if instance.DHCPInterfaceIPv4 != "" {
|
||||
initRebootFlag = true
|
||||
}
|
||||
|
||||
client = vip.NewDHCPv4Client(iface, initRebootFlag, i.DHCPInterfaceIPv4, backoffAttempts, i.dhcpBroadcast)
|
||||
client = vip.NewDHCPv4Client(iface, initRebootFlag, instance.DHCPInterfaceIPv4, backoffAttempts, instance.dhcpBroadcast)
|
||||
|
||||
// Add the client so that we can call it to stop function
|
||||
i.DHCPv4Client = client
|
||||
instance.DHCPv4Client = client
|
||||
|
||||
// Set that DHCPv4 is enabled
|
||||
i.IsDHCPv4 = true
|
||||
instance.IsDHCPv4 = true
|
||||
} else {
|
||||
if i.DHCPInterfaceIPv6 != "" {
|
||||
if instance.DHCPInterfaceIPv6 != "" {
|
||||
initRebootFlag = true
|
||||
}
|
||||
|
||||
client, err = vip.NewDHCPv6Client(iface, parent, initRebootFlag, i.DHCPInterfaceIPv6, backoffAttempts)
|
||||
client, err = vip.NewDHCPv6Client(iface, parent, initRebootFlag, instance.DHCPInterfaceIPv6, backoffAttempts)
|
||||
if err != nil {
|
||||
return fmt.Errorf("unable to create client: %w", err)
|
||||
}
|
||||
|
||||
// Add the client so that we can call it to stop function
|
||||
i.DHCPv6Client = client
|
||||
instance.DHCPv6Client = client
|
||||
|
||||
// Set that DHCPv6 is enabled
|
||||
i.IsDHCPv6 = true
|
||||
instance.IsDHCPv6 = true
|
||||
}
|
||||
|
||||
// Add hostname to dhcp client if annotated
|
||||
if i.DHCPHostname != "" {
|
||||
log.Info("Hostname specified for dhcp lease", "interface", interfaceName, "hostname", i.DHCPHostname)
|
||||
client.WithHostName(i.DHCPHostname)
|
||||
if instance.DHCPHostname != "" {
|
||||
log.Info("Hostname specified for dhcp lease", "interface", interfaceName, "hostname", instance.DHCPHostname)
|
||||
client.WithHostName(instance.DHCPHostname)
|
||||
}
|
||||
|
||||
wg.Go(func() {
|
||||
if err := client.Start(ctx); err != nil {
|
||||
log.Error("[instance] DHCP client error: %w")
|
||||
log.Error("[instance] DHCP client", "error", err)
|
||||
client.Stop()
|
||||
}
|
||||
})
|
||||
|
||||
// Set the name of the interface so that it can be removed on Service deletion
|
||||
i.DHCPInterface = interfaceName
|
||||
i.DHCPInterfaceHwaddr = iface.HardwareAddr.String()
|
||||
instance.DHCPInterface = interfaceName
|
||||
instance.DHCPInterfaceHwaddr = iface.HardwareAddr.String()
|
||||
|
||||
return nil
|
||||
}
|
||||
@@ -736,9 +876,9 @@ func FetchServiceAddresses(s *v1.Service) ([]string, []string) {
|
||||
|
||||
func FindServiceInstance(svc *v1.Service, instances []*Instance) *Instance {
|
||||
log.Debug("finding service", "namespace", svc.Namespace, "name", svc.Name, "UID", svc.UID)
|
||||
for i := range instances {
|
||||
if instances[i].ServiceSnapshot.UID == svc.UID {
|
||||
return instances[i]
|
||||
for index := range instances {
|
||||
if instances[index].UID() == svc.UID {
|
||||
return instances[index]
|
||||
}
|
||||
}
|
||||
log.Debug("instance not found", "namespace", svc.Namespace, "name", svc.Name, "UID", svc.UID)
|
||||
|
||||
62
pkg/instance/instance_bgp_attach_test.go
Normal file
62
pkg/instance/instance_bgp_attach_test.go
Normal file
@@ -0,0 +1,62 @@
|
||||
package instance_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
v1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/arp"
|
||||
"github.com/kube-vip/kube-vip/pkg/instance"
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/networkinterface"
|
||||
"github.com/kube-vip/kube-vip/pkg/route"
|
||||
)
|
||||
|
||||
func TestNewInstance_PropagatesBGPAttachIPToInterface(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
attach bool
|
||||
}{
|
||||
{name: "attach enabled is propagated", attach: true},
|
||||
{name: "attach disabled is propagated", attach: false},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
globalConfig := &kubevip.Config{
|
||||
Interface: "lo",
|
||||
VIPSubnet: "32",
|
||||
EnableBGP: true,
|
||||
BGPAttachIPToInterface: tt.attach,
|
||||
}
|
||||
|
||||
svc := &v1.Service{
|
||||
ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "test-svc",
|
||||
Namespace: "default",
|
||||
Annotations: map[string]string{
|
||||
kubevip.LoadbalancerIPAnnotation: "10.0.1.2",
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
inst, err := instance.NewInstance(context.Background(), svc, globalConfig,
|
||||
networkinterface.NewManager(), arp.NewManager(globalConfig), route.NewManager(),
|
||||
nil, &sync.WaitGroup{})
|
||||
if err != nil {
|
||||
t.Fatalf("NewInstance() error = %v", err)
|
||||
}
|
||||
|
||||
if len(inst.VIPConfigs) != 1 {
|
||||
t.Fatalf("VIPConfigs len = %d, want 1", len(inst.VIPConfigs))
|
||||
}
|
||||
|
||||
if got := inst.VIPConfigs[0].BGPAttachIPToInterface; got != tt.attach {
|
||||
t.Fatalf("BGPAttachIPToInterface = %t, want %t", got, tt.attach)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
65
pkg/instance/instance_test.go
Normal file
65
pkg/instance/instance_test.go
Normal file
@@ -0,0 +1,65 @@
|
||||
package instance
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
)
|
||||
|
||||
func TestUIDUsesImmutableServiceUID(t *testing.T) {
|
||||
serviceUID := types.UID("original-service")
|
||||
instance := &Instance{
|
||||
ServiceUID: serviceUID,
|
||||
ServiceSnapshot: &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
UID: types.UID("replacement-service"),
|
||||
}},
|
||||
}
|
||||
|
||||
if got := instance.UID(); got != serviceUID {
|
||||
t.Fatalf("UID() = %q, want %q", got, serviceUID)
|
||||
}
|
||||
|
||||
instance.ServiceUID = ""
|
||||
if got := instance.UID(); got != "" {
|
||||
t.Fatalf("UID() without ServiceUID = %q, want empty UID", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCleanupStateIsImmutable(t *testing.T) {
|
||||
original := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "service", Namespace: "default", Annotations: map[string]string{kubevip.ServiceLease: "shared"},
|
||||
}, Spec: v1.ServiceSpec{LoadBalancerIP: "192.0.2.10", ExternalTrafficPolicy: v1.ServiceExternalTrafficPolicyTypeCluster}}
|
||||
immutableInfo := serviceCleanupInfo(original)
|
||||
instance := &Instance{
|
||||
ServiceSnapshot: original,
|
||||
ServiceAddresses: []string{"192.0.2.10"},
|
||||
cleanupInfo: &immutableInfo,
|
||||
}
|
||||
instance.ServiceSnapshot = &v1.Service{ObjectMeta: metav1.ObjectMeta{Name: "changed"}}
|
||||
|
||||
cleanupInfo, ok := instance.CleanupInfo()
|
||||
if !ok || cleanupInfo.Namespace != "default" || cleanupInfo.Name != "service" || cleanupInfo.Lease != "shared" ||
|
||||
cleanupInfo.ExternalTrafficPolicy != v1.ServiceExternalTrafficPolicyTypeCluster {
|
||||
t.Fatalf("CleanupInfo() = %+v, %t, want creation-time Service policy", cleanupInfo, ok)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTransferLinkAttachmentOwnershipConcurrent(t *testing.T) {
|
||||
target := &Instance{IsVLAN: true, VLANInterface: "eth0.42"}
|
||||
var wg sync.WaitGroup
|
||||
for range 100 {
|
||||
wg.Go(func() {
|
||||
if !transferLinkAttachmentOwnership("eth0.42", []*Instance{target}, true) {
|
||||
t.Error("transferLinkAttachmentOwnership() did not find target")
|
||||
}
|
||||
})
|
||||
}
|
||||
wg.Wait()
|
||||
if !target.vlanOwned.Load() {
|
||||
t.Fatal("target did not receive VLAN ownership")
|
||||
}
|
||||
}
|
||||
102
pkg/instance/links_linux_test.go
Normal file
102
pkg/instance/links_linux_test.go
Normal file
@@ -0,0 +1,102 @@
|
||||
//go:build linux
|
||||
|
||||
package instance
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"os"
|
||||
"runtime"
|
||||
"testing"
|
||||
|
||||
"github.com/vishvananda/netlink"
|
||||
"github.com/vishvananda/netns"
|
||||
)
|
||||
|
||||
// requireNetworkNamespaces makes the privileged CI job fail instead of silently
|
||||
// skipping when it cannot enter a network namespace.
|
||||
var requireNetworkNamespaces = os.Getenv("KUBE_VIP_REQUIRE_NETNS") != ""
|
||||
|
||||
func TestCleanupLinkAttachmentsOnlyDeletesOwnedVLAN(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
preexists bool
|
||||
inUse bool
|
||||
}{
|
||||
{name: "owned VLAN", preexists: false},
|
||||
{name: "adopted VLAN", preexists: true},
|
||||
{name: "owned VLAN used by another Service", inUse: true},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
runtime.LockOSThread()
|
||||
defer runtime.UnlockOSThread()
|
||||
|
||||
originalNamespace, err := netns.Get()
|
||||
if err != nil {
|
||||
t.Fatalf("getting current network namespace: %v", err)
|
||||
}
|
||||
defer originalNamespace.Close()
|
||||
testNamespace, err := netns.New()
|
||||
if err != nil {
|
||||
if requireNetworkNamespaces {
|
||||
t.Fatalf("creating isolated network namespace: %v", err)
|
||||
}
|
||||
t.Skipf("creating isolated network namespace: %v", err)
|
||||
}
|
||||
defer testNamespace.Close()
|
||||
defer func() {
|
||||
if err := netns.Set(originalNamespace); err != nil {
|
||||
t.Errorf("restoring network namespace: %v", err)
|
||||
}
|
||||
}()
|
||||
|
||||
parent := &netlink.Dummy{LinkAttrs: netlink.LinkAttrs{Name: "kvattach0"}}
|
||||
if err := netlink.LinkAdd(parent); err != nil {
|
||||
t.Fatalf("creating parent interface: %v", err)
|
||||
}
|
||||
if err := netlink.LinkSetUp(parent); err != nil {
|
||||
t.Fatalf("bringing up parent interface: %v", err)
|
||||
}
|
||||
if test.preexists {
|
||||
vlan := &netlink.Vlan{LinkAttrs: netlink.LinkAttrs{Name: "kvattach0.42", ParentIndex: parent.Attrs().Index}, VlanId: 42}
|
||||
if err := netlink.LinkAdd(vlan); err != nil {
|
||||
t.Fatalf("creating existing VLAN: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
instance := &Instance{}
|
||||
if err := instance.addVLAN(parent.Attrs().Name, 42); err != nil {
|
||||
t.Fatalf("adding VLAN attachment: %v", err)
|
||||
}
|
||||
if instance.vlanOwned.Load() == test.preexists {
|
||||
t.Fatalf("vlanOwned = %t, want %t", instance.vlanOwned.Load(), !test.preexists)
|
||||
}
|
||||
var remaining []*Instance
|
||||
if test.inUse {
|
||||
remaining = []*Instance{{IsVLAN: true, VLANInterface: instance.VLANInterface}}
|
||||
}
|
||||
if err := instance.CleanupLinkAttachments(remaining...); err != nil {
|
||||
t.Fatalf("cleaning attachments: %v", err)
|
||||
}
|
||||
if test.inUse {
|
||||
if !remaining[0].vlanOwned.Load() {
|
||||
t.Fatal("remaining Service did not receive VLAN cleanup ownership")
|
||||
}
|
||||
if _, err := netlink.LinkByName("kvattach0.42"); err != nil {
|
||||
t.Fatalf("VLAN was removed while a Service still used it: %v", err)
|
||||
}
|
||||
if err := remaining[0].CleanupLinkAttachments(); err != nil {
|
||||
t.Fatalf("cleaning transferred attachment: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
_, err = netlink.LinkByName("kvattach0.42")
|
||||
var notFound netlink.LinkNotFoundError
|
||||
if test.preexists && err != nil {
|
||||
t.Fatalf("adopted VLAN was removed: %v", err)
|
||||
}
|
||||
if !test.preexists && !errors.As(err, ¬Found) {
|
||||
t.Fatalf("owned VLAN remains after cleanup: %v", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -18,7 +18,6 @@ import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"io"
|
||||
"net"
|
||||
"os/exec"
|
||||
"regexp"
|
||||
"strconv"
|
||||
@@ -87,21 +86,6 @@ type IPTables struct {
|
||||
|
||||
nftables bool
|
||||
}
|
||||
|
||||
// Stat represents a structured statistic entry.
|
||||
type Stat struct {
|
||||
Packets uint64 `json:"pkts"`
|
||||
Bytes uint64 `json:"bytes"`
|
||||
Target string `json:"target"`
|
||||
Protocol string `json:"prot"`
|
||||
Opt string `json:"opt"`
|
||||
Input string `json:"in"`
|
||||
Output string `json:"out"`
|
||||
Source *net.IPNet `json:"source"`
|
||||
Destination *net.IPNet `json:"destination"`
|
||||
Options string `json:"options"`
|
||||
}
|
||||
|
||||
type Option func(*IPTables)
|
||||
|
||||
func IPFamily(proto Protocol) Option {
|
||||
@@ -251,16 +235,6 @@ func (ipt *IPTables) DeleteIfExists(table, chain string, rulespec ...string) err
|
||||
return err
|
||||
}
|
||||
|
||||
// List rules in specified table/chain
|
||||
func (ipt *IPTables) ListByID(table, chain string, id int) (string, error) {
|
||||
args := []string{"-t", table, "-S", chain, strconv.Itoa(id)}
|
||||
rule, err := ipt.executeList(args)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return rule[0], nil
|
||||
}
|
||||
|
||||
// List rules in specified table/chain
|
||||
func (ipt *IPTables) List(table, chain string) ([]string, error) {
|
||||
args := []string{"-t", table, "-S", chain}
|
||||
@@ -313,129 +287,6 @@ func (ipt *IPTables) ChainExists(table, chain string) (bool, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// Stats lists rules including the byte and packet counts
|
||||
func (ipt *IPTables) Stats(table, chain string) ([][]string, error) {
|
||||
args := []string{"-t", table, "-L", chain, "-n", "-v", "-x"}
|
||||
lines, err := ipt.executeList(args)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
appendSubnet := func(addr string) string {
|
||||
if strings.IndexByte(addr, byte('/')) < 0 {
|
||||
if strings.IndexByte(addr, '.') < 0 {
|
||||
return addr + "/128"
|
||||
}
|
||||
return addr + "/32"
|
||||
}
|
||||
return addr
|
||||
}
|
||||
|
||||
ipv6 := ipt.proto == ProtocolIPv6
|
||||
|
||||
rows := [][]string{}
|
||||
for i, line := range lines {
|
||||
// Skip over chain name and field header
|
||||
if i < 2 {
|
||||
continue
|
||||
}
|
||||
|
||||
// Fields:
|
||||
// 0=pkts 1=bytes 2=target 3=prot 4=opt 5=in 6=out 7=source 8=destination 9=options
|
||||
line = strings.TrimSpace(line)
|
||||
fields := strings.Fields(line)
|
||||
|
||||
// The ip6tables verbose output cannot be naively split due to the default "opt"
|
||||
// field containing 2 single spaces.
|
||||
if ipv6 {
|
||||
// Check if field 6 is "opt" or "source" address
|
||||
dest := fields[6]
|
||||
ip, _, _ := net.ParseCIDR(dest)
|
||||
if ip == nil {
|
||||
ip = net.ParseIP(dest)
|
||||
}
|
||||
|
||||
// If we detected a CIDR or IP, the "opt" field is empty.. insert it.
|
||||
if ip != nil {
|
||||
f := []string{}
|
||||
f = append(f, fields[:4]...)
|
||||
f = append(f, " ") // Empty "opt" field for ip6tables
|
||||
f = append(f, fields[4:]...)
|
||||
fields = f
|
||||
}
|
||||
}
|
||||
|
||||
// Adjust "source" and "destination" to include netmask, to match regular
|
||||
// List output
|
||||
fields[7] = appendSubnet(fields[7])
|
||||
fields[8] = appendSubnet(fields[8])
|
||||
|
||||
// Combine "options" fields 9... into a single space-delimited field.
|
||||
options := fields[9:]
|
||||
fields = fields[:9]
|
||||
fields = append(fields, strings.Join(options, " "))
|
||||
rows = append(rows, fields)
|
||||
}
|
||||
return rows, nil
|
||||
}
|
||||
|
||||
// ParseStat parses a single statistic row into a Stat struct. The input should
|
||||
// be a string slice that is returned from calling the Stat method.
|
||||
func (ipt *IPTables) ParseStat(stat []string) (parsed Stat, err error) {
|
||||
// For forward-compatibility, expect at least 10 fields in the stat
|
||||
if len(stat) < 10 {
|
||||
return parsed, fmt.Errorf("stat contained fewer fields than expected")
|
||||
}
|
||||
|
||||
// Convert the fields that are not plain strings
|
||||
parsed.Packets, err = strconv.ParseUint(stat[0], 0, 64)
|
||||
if err != nil {
|
||||
return parsed, fmt.Errorf(err.Error(), "could not parse packets")
|
||||
}
|
||||
parsed.Bytes, err = strconv.ParseUint(stat[1], 0, 64)
|
||||
if err != nil {
|
||||
return parsed, fmt.Errorf(err.Error(), "could not parse bytes")
|
||||
}
|
||||
_, parsed.Source, err = net.ParseCIDR(stat[7])
|
||||
if err != nil {
|
||||
return parsed, fmt.Errorf(err.Error(), "could not parse source")
|
||||
}
|
||||
_, parsed.Destination, err = net.ParseCIDR(stat[8])
|
||||
if err != nil {
|
||||
return parsed, fmt.Errorf(err.Error(), "could not parse destination")
|
||||
}
|
||||
|
||||
// Put the fields that are strings
|
||||
parsed.Target = stat[2]
|
||||
parsed.Protocol = stat[3]
|
||||
parsed.Opt = stat[4]
|
||||
parsed.Input = stat[5]
|
||||
parsed.Output = stat[6]
|
||||
parsed.Options = stat[9]
|
||||
|
||||
return parsed, nil
|
||||
}
|
||||
|
||||
// StructuredStats returns statistics as structured data which may be further
|
||||
// parsed and marshaled.
|
||||
func (ipt *IPTables) StructuredStats(table, chain string) ([]Stat, error) {
|
||||
rawStats, err := ipt.Stats(table, chain)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
structStats := []Stat{}
|
||||
for _, rawStat := range rawStats {
|
||||
stat, err := ipt.ParseStat(rawStat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
structStats = append(structStats, stat)
|
||||
}
|
||||
|
||||
return structStats, nil
|
||||
}
|
||||
|
||||
func (ipt *IPTables) executeList(args []string) ([]string, error) {
|
||||
var stdout bytes.Buffer
|
||||
if err := ipt.runWithOutput(args, &stdout); err != nil {
|
||||
|
||||
@@ -77,6 +77,9 @@ const (
|
||||
// Name of the service lease object
|
||||
ServiceLease = "kube-vip.io/leaseName"
|
||||
|
||||
// Versioned kube-vip ownership metadata stored on Kubernetes election Leases
|
||||
LeaseVIPs = "kube-vip.io/lease-vips"
|
||||
|
||||
// Forces kube-vip to use per service election for this particular service
|
||||
ForcePerServiceElection = "kube-vip.io/forcePerServiceElection"
|
||||
|
||||
|
||||
@@ -2,7 +2,6 @@ package kubevip
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"net"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
@@ -234,27 +233,44 @@ func ParseBGPPeerConfig(config string) (bgpPeers []BGPPeer, err error) {
|
||||
|
||||
func (p *BGPPeer) FindMpbgpAddresses(ap *api.Peer, server *BGPConfig) (string, string, error) {
|
||||
var ipv4Address, ipv6Address string
|
||||
switch p.MpbgpNexthop {
|
||||
|
||||
mode := server.MpbgpNexthop
|
||||
if p.MpbgpNexthop != "" {
|
||||
mode = p.MpbgpNexthop
|
||||
}
|
||||
|
||||
switch mode {
|
||||
case "fixed":
|
||||
ap.Transport.LocalAddress = server.SourceIP
|
||||
if p.MpbgpIPv4 == "" && p.MpbgpIPv6 == "" {
|
||||
return "", "", fmt.Errorf("to use MP-BGP with fixed address at least one IPv4 or IPv6 address has to be provided [current - IPv4: %s, IPv6: %s]",
|
||||
p.MpbgpIPv4, p.MpbgpIPv6)
|
||||
}
|
||||
|
||||
ipv4 := server.MpbgpIPv4
|
||||
if p.MpbgpIPv4 != "" {
|
||||
if net.ParseIP(p.MpbgpIPv4) == nil {
|
||||
return "", "", fmt.Errorf("provided address '%s' is not a valid IPv4 address", p.MpbgpIPv4)
|
||||
ipv4 = p.MpbgpIPv4
|
||||
}
|
||||
|
||||
ipv6 := server.MpbgpIPv6
|
||||
if p.MpbgpIPv6 != "" {
|
||||
ipv6 = p.MpbgpIPv6
|
||||
}
|
||||
|
||||
if ipv4 == "" && ipv6 == "" {
|
||||
return "", "", fmt.Errorf("to use MP-BGP with fixed address at least one IPv4 or IPv6 address has to be provided [current - IPv4: %s, IPv6: %s]",
|
||||
ipv4, ipv6)
|
||||
}
|
||||
|
||||
if ipv4 != "" {
|
||||
if !utils.IsIPv4(ipv4) {
|
||||
return "", "", fmt.Errorf("provided address '%s' is not a valid IPv4 address", ipv4)
|
||||
}
|
||||
}
|
||||
if p.MpbgpIPv6 != "" {
|
||||
if net.ParseIP(p.MpbgpIPv6) == nil {
|
||||
return "", "", fmt.Errorf("provided address '%s' is not a valid IPv6 address", p.MpbgpIPv6)
|
||||
if ipv6 != "" {
|
||||
if !utils.IsIPv6(ipv6) {
|
||||
return "", "", fmt.Errorf("provided address '%s' is not a valid IPv6 address", ipv6)
|
||||
}
|
||||
}
|
||||
|
||||
ipv4Address = p.MpbgpIPv4
|
||||
ipv6Address = p.MpbgpIPv6
|
||||
ipv4Address = ipv4
|
||||
ipv6Address = ipv6
|
||||
case "auto_sourceip":
|
||||
ap.Transport.LocalAddress = server.SourceIP
|
||||
|
||||
@@ -297,7 +313,7 @@ func (p *BGPPeer) FindMpbgpAddresses(ap *api.Peer, server *BGPConfig) (string, s
|
||||
return "", "", fmt.Errorf("failed to get non link-local IPv6 address: %v", err)
|
||||
}
|
||||
default:
|
||||
return "", "", fmt.Errorf("option %s for MP-BPG nexthop is not supported", server.MpbgpNexthop)
|
||||
return "", "", fmt.Errorf("option %q for MP-BPG nexthop is not supported", mode)
|
||||
}
|
||||
|
||||
return ipv4Address, ipv6Address, nil
|
||||
|
||||
38
pkg/kubevip/config_bgp_family_test.go
Normal file
38
pkg/kubevip/config_bgp_family_test.go
Normal file
@@ -0,0 +1,38 @@
|
||||
package kubevip
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
api "github.com/osrg/gobgp/v4/api"
|
||||
)
|
||||
|
||||
func TestFindMpbgpAddressesRejectsFixedAddressFamilyMismatches(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
peer BGPPeer
|
||||
}{
|
||||
{
|
||||
name: "IPv6 value in IPv4 field",
|
||||
peer: BGPPeer{
|
||||
MpbgpNexthop: "fixed",
|
||||
MpbgpIPv4: "2001:db8::20",
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "IPv4 value in IPv6 field",
|
||||
peer: BGPPeer{
|
||||
MpbgpNexthop: "fixed",
|
||||
MpbgpIPv6: "192.0.2.20",
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
_, _, err := tt.peer.FindMpbgpAddresses(&api.Peer{Transport: &api.Transport{}}, &BGPConfig{})
|
||||
if err == nil {
|
||||
t.Fatal("FindMpbgpAddresses() error = nil, want address-family error")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -406,6 +406,16 @@ func ParseEnvironment(c *Config) error {
|
||||
c.CleanRoutingTable = b
|
||||
}
|
||||
|
||||
// Skip Duplicate Address Detection when adding the VIP address
|
||||
env = os.Getenv(vipSkipDAD)
|
||||
if env != "" {
|
||||
b, err := strconv.ParseBool(env)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
c.SkipDAD = b
|
||||
}
|
||||
|
||||
// DNS mode
|
||||
env = os.Getenv(dnsMode)
|
||||
if env != "" {
|
||||
@@ -456,6 +466,15 @@ func ParseEnvironment(c *Config) error {
|
||||
c.EnableBGP = b
|
||||
}
|
||||
|
||||
env = os.Getenv(bgpAttachIPToInterface)
|
||||
if env != "" {
|
||||
b, err := strconv.ParseBool(env)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
c.BGPAttachIPToInterface = b
|
||||
}
|
||||
|
||||
// BGP Router interface determines an interface that we can use to find an address for
|
||||
env = os.Getenv(bgpRouterInterface)
|
||||
if env != "" {
|
||||
@@ -897,6 +916,9 @@ func mergeConfigValues(baseConfig, fileConfig *Config) {
|
||||
if !baseConfig.EnableBGP && fileConfig.EnableBGP {
|
||||
baseConfig.EnableBGP = fileConfig.EnableBGP
|
||||
}
|
||||
if !baseConfig.BGPAttachIPToInterface && fileConfig.BGPAttachIPToInterface {
|
||||
baseConfig.BGPAttachIPToInterface = fileConfig.BGPAttachIPToInterface
|
||||
}
|
||||
if !baseConfig.EnableWireguard && fileConfig.EnableWireguard {
|
||||
baseConfig.EnableWireguard = fileConfig.EnableWireguard
|
||||
}
|
||||
|
||||
41
pkg/kubevip/config_environment_test.go
Normal file
41
pkg/kubevip/config_environment_test.go
Normal file
@@ -0,0 +1,41 @@
|
||||
package kubevip
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestParseEnvironmentSkipDAD(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
value string
|
||||
want bool
|
||||
wantErr bool
|
||||
}{
|
||||
{name: "unset keeps default false", value: "", want: false},
|
||||
{name: "true enables", value: "true", want: true},
|
||||
{name: "false disables", value: "false", want: false},
|
||||
{name: "garbage errors", value: "not-a-bool", wantErr: true},
|
||||
}
|
||||
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if tc.value != "" {
|
||||
t.Setenv(vipSkipDAD, tc.value)
|
||||
}
|
||||
c := &Config{}
|
||||
err := ParseEnvironment(c)
|
||||
if tc.wantErr {
|
||||
if err == nil {
|
||||
t.Fatal("expected an error, got nil")
|
||||
}
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if c.SkipDAD != tc.want {
|
||||
t.Fatalf("SkipDAD = %v, want %v", c.SkipDAD, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -105,6 +105,8 @@ const (
|
||||
|
||||
// bgpEnable defines if BGP should be enabled
|
||||
bgpEnable = "bgp_enable"
|
||||
// bgpAttachIPToInterface defines if BGP service VIPs should be assigned to the configured interface
|
||||
bgpAttachIPToInterface = "bgp_attach_ip_to_interface"
|
||||
// bgpRouterID defines the routerID for the BGP server
|
||||
bgpRouterID = "bgp_routerid"
|
||||
// bgpRouterInterface defines the interface that we can find the address for
|
||||
@@ -179,6 +181,9 @@ const (
|
||||
// vipCleanRoutingTable - defines if routing table will be cleaned of redundant routes on kube-vip's start
|
||||
vipCleanRoutingTable = "vip_cleanroutingtable" //nolint
|
||||
|
||||
// vipSkipDAD - defines if Duplicate Address Detection is skipped when adding the VIP address (IFA_F_NODAD)
|
||||
vipSkipDAD = "vip_skipdad" //nolint
|
||||
|
||||
// cpNamespace defines the namespace the control plane pods will run in
|
||||
cpNamespace = "cp_namespace"
|
||||
|
||||
|
||||
@@ -125,6 +125,13 @@ func GenerateRole(c *Config, role bool) *applyRbacV1.RoleApplyConfiguration {
|
||||
},
|
||||
},
|
||||
}
|
||||
if !role {
|
||||
newManifest.Rules = append(newManifest.Rules, applyRbacV1.PolicyRuleApplyConfiguration{
|
||||
APIGroups: []string{"networking.k8s.io"},
|
||||
Resources: []string{"servicecidrs"},
|
||||
Verbs: []string{"list", "get", "watch"},
|
||||
})
|
||||
}
|
||||
return newManifest
|
||||
}
|
||||
|
||||
@@ -476,6 +483,12 @@ func generatePodSpec(c *Config, image, imageVersion string, inCluster bool) (*co
|
||||
Value: strconv.FormatBool(c.EnableBGP),
|
||||
},
|
||||
}
|
||||
if c.BGPAttachIPToInterface {
|
||||
bgp = append(bgp, corev1.EnvVar{
|
||||
Name: bgpAttachIPToInterface,
|
||||
Value: strconv.FormatBool(c.BGPAttachIPToInterface),
|
||||
})
|
||||
}
|
||||
newEnvironment = append(newEnvironment, bgp...)
|
||||
}
|
||||
|
||||
|
||||
@@ -2,10 +2,38 @@ package kubevip
|
||||
|
||||
import (
|
||||
"os"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
applyRbacV1 "k8s.io/client-go/applyconfigurations/rbac/v1"
|
||||
)
|
||||
|
||||
func TestGenerateRoleServiceCIDRAccess(t *testing.T) {
|
||||
clusterRole := GenerateRole(&Config{}, false)
|
||||
if !hasServiceCIDRRule(clusterRole) {
|
||||
t.Fatal("generated ClusterRole is missing ServiceCIDR access")
|
||||
}
|
||||
|
||||
role := GenerateRole(&Config{ServiceNamespace: "kube-vip"}, true)
|
||||
if hasServiceCIDRRule(role) {
|
||||
t.Fatal("generated namespaced Role contains ineffective ServiceCIDR access")
|
||||
}
|
||||
}
|
||||
|
||||
func hasServiceCIDRRule(role *applyRbacV1.RoleApplyConfiguration) bool {
|
||||
for _, rule := range role.Rules {
|
||||
if slices.Contains(rule.APIGroups, "networking.k8s.io") &&
|
||||
slices.Contains(rule.Resources, "servicecidrs") &&
|
||||
slices.Contains(rule.Verbs, "get") &&
|
||||
slices.Contains(rule.Verbs, "list") &&
|
||||
slices.Contains(rule.Verbs, "watch") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func TestParseEnvironment(t *testing.T) {
|
||||
|
||||
tests := []struct {
|
||||
@@ -54,6 +82,35 @@ func TestParseEnvironmentInstanceName(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseEnvironmentBGPAttachIPToInterface(t *testing.T) {
|
||||
t.Setenv(bgpAttachIPToInterface, "true")
|
||||
|
||||
config := &Config{}
|
||||
if err := ParseEnvironment(config); err != nil {
|
||||
t.Fatalf("ParseEnvironment() error = %v", err)
|
||||
}
|
||||
if !config.BGPAttachIPToInterface {
|
||||
t.Fatal("BGPAttachIPToInterface = false, want true")
|
||||
}
|
||||
}
|
||||
|
||||
func TestGeneratePodSpecBGPAttachIPToInterface(t *testing.T) {
|
||||
pod, err := generatePodSpec(&Config{
|
||||
EnableBGP: true,
|
||||
BGPAttachIPToInterface: true,
|
||||
}, "ghcr.io/kube-vip/kube-vip", "v0.0.0", true)
|
||||
if err != nil {
|
||||
t.Fatalf("generatePodSpec() error = %v", err)
|
||||
}
|
||||
|
||||
for _, env := range pod.Spec.Containers[0].Env {
|
||||
if env.Name == bgpAttachIPToInterface && env.Value == "true" {
|
||||
return
|
||||
}
|
||||
}
|
||||
t.Fatalf("%s=true is missing from generated pod environment", bgpAttachIPToInterface)
|
||||
}
|
||||
|
||||
func TestGeneratePodSpecInstanceName(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
|
||||
@@ -11,6 +11,9 @@ type Config struct {
|
||||
// EnableBGP, will use BGP to advertise the VIP address
|
||||
EnableBGP bool `yaml:"enableBGP"`
|
||||
|
||||
// BGPAttachIPToInterface assigns BGP-advertised service VIPs to the configured interface
|
||||
BGPAttachIPToInterface bool `yaml:"bgpAttachIPToInterface"`
|
||||
|
||||
// EnableWireguard, will use wireguard to advertise the VIP address
|
||||
EnableWireguard bool `yaml:"enableWireguard"`
|
||||
|
||||
@@ -139,6 +142,9 @@ type Config struct {
|
||||
// Clean routing table of redundant routes on start
|
||||
CleanRoutingTable bool `yaml:"cleanRoutingTable"`
|
||||
|
||||
// Skip Duplicate Address Detection when adding the VIP address (IFA_F_NODAD)
|
||||
SkipDAD bool `yaml:"skipDAD"`
|
||||
|
||||
// BGP Configuration
|
||||
BGPConfig BGPConfig
|
||||
BGPPeerConfig BGPPeer
|
||||
|
||||
@@ -2,6 +2,7 @@ package kubevip
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"net/url"
|
||||
"strings"
|
||||
)
|
||||
@@ -24,10 +25,23 @@ func (c *Config) Validate() error {
|
||||
if err := validateInstanceName(c.InstanceName); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := validateRoutingProtocol(c.RoutingProtocol); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// validateRoutingProtocol rejects values the kernel cannot represent: netlink
|
||||
// carries the address and route protocol in a single byte, so a larger value is
|
||||
// silently truncated on the wire and never matches on readback.
|
||||
func validateRoutingProtocol(protocol int) error {
|
||||
if protocol < 0 || protocol > math.MaxUint8 {
|
||||
return fmt.Errorf("routingProtocol %d is out of range, must be between 0 and %d", protocol, math.MaxUint8)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func validateInstanceName(name string) error {
|
||||
if name == "" {
|
||||
return nil
|
||||
|
||||
@@ -62,6 +62,30 @@ func TestValidate_InstanceName(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidate_RoutingProtocol(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
protocol int
|
||||
wantErr bool
|
||||
}{
|
||||
{name: "unset", protocol: 0, wantErr: false},
|
||||
{name: "kube-vip default", protocol: 248, wantErr: false},
|
||||
{name: "maximum byte value", protocol: 255, wantErr: false},
|
||||
{name: "truncated on the wire", protocol: 256, wantErr: true},
|
||||
{name: "negative", protocol: -1, wantErr: true},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
config := &Config{RoutingProtocol: tt.protocol}
|
||||
err := config.Validate()
|
||||
if (err != nil) != tt.wantErr {
|
||||
t.Fatalf("Validate() error = %v, wantErr %t", err, tt.wantErr)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestInstanceNameLimitReservesNftablesPrefixAndFamilySuffix(t *testing.T) {
|
||||
name := strings.Repeat("a", instanceNameMaxLength)
|
||||
if got := len(egressNftablesTablePrefix + name + egressNftablesTableSuffix); got != nftablesNameMaxLength {
|
||||
|
||||
102
pkg/kubevip/lease_annotations.go
Normal file
102
pkg/kubevip/lease_annotations.go
Normal file
@@ -0,0 +1,102 @@
|
||||
package kubevip
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/netip"
|
||||
"slices"
|
||||
"strings"
|
||||
)
|
||||
|
||||
const LeaseVIPsVersion = "v1"
|
||||
|
||||
type LeaseVIPsValue struct {
|
||||
Version string `json:"version"`
|
||||
InstanceName string `json:"instance_name"`
|
||||
IFAProto int `json:"ifa_proto"`
|
||||
VIPs []LeaseVIP `json:"vips"`
|
||||
}
|
||||
|
||||
type LeaseVIP struct {
|
||||
Index int `json:"index"`
|
||||
Value string `json:"value"`
|
||||
}
|
||||
|
||||
func WithLeaseVIPs(annotations map[string]string, instanceName string, ifaProto int, vips []string) (map[string]string, error) {
|
||||
result := make(map[string]string, len(annotations)+1)
|
||||
for key, value := range annotations {
|
||||
result[key] = value
|
||||
}
|
||||
|
||||
encoded, err := json.Marshal(LeaseVIPsValue{
|
||||
Version: LeaseVIPsVersion,
|
||||
InstanceName: instanceName,
|
||||
IFAProto: ifaProto,
|
||||
VIPs: normalizeLeaseVIPs(vips),
|
||||
})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("encode %s annotation: %w", LeaseVIPs, err)
|
||||
}
|
||||
result[LeaseVIPs] = string(encoded)
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func ParseLeaseVIPs(value string) (LeaseVIPsValue, error) {
|
||||
var parsed LeaseVIPsValue
|
||||
if err := json.Unmarshal([]byte(value), &parsed); err != nil {
|
||||
return LeaseVIPsValue{}, fmt.Errorf("decode %s annotation: %w", LeaseVIPs, err)
|
||||
}
|
||||
if parsed.Version != LeaseVIPsVersion {
|
||||
return LeaseVIPsValue{}, fmt.Errorf("unsupported %s annotation version %q", LeaseVIPs, parsed.Version)
|
||||
}
|
||||
for index, vip := range parsed.VIPs {
|
||||
if vip.Index != index {
|
||||
return LeaseVIPsValue{}, fmt.Errorf("invalid %s VIP index %d at position %d", LeaseVIPs, vip.Index, index)
|
||||
}
|
||||
address, err := parseLeaseVIP(vip.Value)
|
||||
if err != nil {
|
||||
return LeaseVIPsValue{}, fmt.Errorf("invalid %s VIP at index %d: %w", LeaseVIPs, vip.Index, err)
|
||||
}
|
||||
parsed.VIPs[index].Value = address.String()
|
||||
}
|
||||
return parsed, nil
|
||||
}
|
||||
|
||||
func normalizeLeaseVIPs(values []string) []LeaseVIP {
|
||||
unique := make(map[netip.Addr]struct{})
|
||||
addresses := make([]netip.Addr, 0, len(values))
|
||||
for _, value := range values {
|
||||
for candidate := range strings.SplitSeq(value, ",") {
|
||||
candidate = strings.TrimSpace(candidate)
|
||||
address, err := parseLeaseVIP(candidate)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if _, exists := unique[address]; exists {
|
||||
continue
|
||||
}
|
||||
unique[address] = struct{}{}
|
||||
addresses = append(addresses, address)
|
||||
}
|
||||
}
|
||||
// Sorting keeps the annotation byte-identical however callers happen to order VIPs.
|
||||
slices.SortFunc(addresses, netip.Addr.Compare)
|
||||
|
||||
result := make([]LeaseVIP, 0, len(addresses))
|
||||
for _, address := range addresses {
|
||||
result = append(result, LeaseVIP{Index: len(result), Value: address.String()})
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
func parseLeaseVIP(value string) (netip.Addr, error) {
|
||||
address, err := netip.ParseAddr(value)
|
||||
if err == nil {
|
||||
return address.Unmap(), nil
|
||||
}
|
||||
prefix, prefixErr := netip.ParsePrefix(value)
|
||||
if prefixErr != nil {
|
||||
return netip.Addr{}, fmt.Errorf("parse address %q: %w", value, err)
|
||||
}
|
||||
return prefix.Addr().Unmap(), nil
|
||||
}
|
||||
78
pkg/kubevip/lease_annotations_test.go
Normal file
78
pkg/kubevip/lease_annotations_test.go
Normal file
@@ -0,0 +1,78 @@
|
||||
package kubevip
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestWithLeaseVIPsEncodesVersionedInstanceOwnership(t *testing.T) {
|
||||
base := map[string]string{"example.test/preserved": "true", LeaseVIPs: "stale"}
|
||||
annotations, err := WithLeaseVIPs(base, "release_a", 248, []string{
|
||||
"2001:db8::10/128", "192.0.2.10", "192.0.2.10/32", "api.example.test",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("WithLeaseVIPs() error = %v", err)
|
||||
}
|
||||
if annotations["example.test/preserved"] != "true" {
|
||||
t.Fatal("WithLeaseVIPs() dropped an existing annotation")
|
||||
}
|
||||
if base[LeaseVIPs] != "stale" {
|
||||
t.Fatal("WithLeaseVIPs() mutated the input annotations")
|
||||
}
|
||||
|
||||
value, err := ParseLeaseVIPs(annotations[LeaseVIPs])
|
||||
if err != nil {
|
||||
t.Fatalf("ParseLeaseVIPs() error = %v", err)
|
||||
}
|
||||
if value.Version != LeaseVIPsVersion || value.InstanceName != "release_a" || value.IFAProto != 248 {
|
||||
t.Fatalf("Lease VIP metadata = %+v", value)
|
||||
}
|
||||
if len(value.VIPs) != 2 ||
|
||||
value.VIPs[0] != (LeaseVIP{Index: 0, Value: "192.0.2.10"}) ||
|
||||
value.VIPs[1] != (LeaseVIP{Index: 1, Value: "2001:db8::10"}) {
|
||||
t.Fatalf("Lease VIPs = %v, want indexed VIPs in canonical address order", value.VIPs)
|
||||
}
|
||||
}
|
||||
|
||||
// The annotation is rewritten whenever a node starts campaigning, so the encoding
|
||||
// has to be stable even when callers collect the same VIPs in a different order.
|
||||
func TestWithLeaseVIPsIsIndependentOfInputOrder(t *testing.T) {
|
||||
first, err := WithLeaseVIPs(nil, "release_a", 248, []string{
|
||||
"2001:db8::10", "192.0.2.10", "10.0.0.2", "10.0.0.10",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("WithLeaseVIPs() error = %v", err)
|
||||
}
|
||||
second, err := WithLeaseVIPs(nil, "release_a", 248, []string{
|
||||
"10.0.0.10", "192.0.2.10", "2001:db8::10", "10.0.0.2",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("WithLeaseVIPs() error = %v", err)
|
||||
}
|
||||
if first[LeaseVIPs] != second[LeaseVIPs] {
|
||||
t.Fatalf("annotation changed with input order:\n%s\n%s", first[LeaseVIPs], second[LeaseVIPs])
|
||||
}
|
||||
|
||||
value, err := ParseLeaseVIPs(first[LeaseVIPs])
|
||||
if err != nil {
|
||||
t.Fatalf("ParseLeaseVIPs() error = %v", err)
|
||||
}
|
||||
want := []string{"10.0.0.2", "10.0.0.10", "192.0.2.10", "2001:db8::10"}
|
||||
if len(value.VIPs) != len(want) {
|
||||
t.Fatalf("Lease VIPs = %v, want %v", value.VIPs, want)
|
||||
}
|
||||
for index, address := range want {
|
||||
if value.VIPs[index] != (LeaseVIP{Index: index, Value: address}) {
|
||||
t.Fatalf("Lease VIPs = %v, want %v", value.VIPs, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseLeaseVIPsRejectsUnknownVersion(t *testing.T) {
|
||||
if _, err := ParseLeaseVIPs(`{"version":"v2","instance_name":"release_a","ifa_proto":248,"vips":[]}`); err == nil {
|
||||
t.Fatal("ParseLeaseVIPs() accepted an unknown version")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseLeaseVIPsRejectsOutOfOrderIndexes(t *testing.T) {
|
||||
if _, err := ParseLeaseVIPs(`{"version":"v1","instance_name":"release_a","ifa_proto":248,"vips":[{"index":1,"value":"192.0.2.10"}]}`); err == nil {
|
||||
t.Fatal("ParseLeaseVIPs() accepted an out-of-order VIP index")
|
||||
}
|
||||
}
|
||||
@@ -26,18 +26,41 @@ func NewManager() *Manager {
|
||||
}
|
||||
}
|
||||
|
||||
// Add adds lease to the manager.
|
||||
// It returns three values:
|
||||
// - lease for the object
|
||||
// - isNewObject, which reports if it is a new object that is being handled
|
||||
// - isSharedLease, which is true if object shares the lease with another object
|
||||
// If object is new but not shared, we should start leaderelection and sync it
|
||||
// If object is new and shared, we should only sync it as the leaderelection should be already handled
|
||||
// If object is not new we should do nothing
|
||||
// Add creates or retrieves the lease identified by id.
|
||||
func (m *Manager) Add(ctx context.Context, id ID) *Lease {
|
||||
m.lock.Lock()
|
||||
defer m.lock.Unlock()
|
||||
if _, exists := m.leases[id.NamespacedName()]; !exists {
|
||||
return m.addLocked(ctx, id)
|
||||
}
|
||||
|
||||
// Acquire creates or retrieves a lease and atomically registers objectName as a
|
||||
// member. The returned bool reports whether this object was newly registered.
|
||||
func (m *Manager) Acquire(ctx context.Context, id ID, objectName string) (*Lease, bool) {
|
||||
m.lock.Lock()
|
||||
defer m.lock.Unlock()
|
||||
|
||||
lease := m.addLocked(ctx, id)
|
||||
return lease, lease.Add(objectName)
|
||||
}
|
||||
|
||||
// Claim atomically registers objectName against an existing lease. It returns
|
||||
// nil when the lease was retired before the caller could join it.
|
||||
func (m *Manager) Claim(id ID, objectName string) (*Lease, bool) {
|
||||
m.lock.Lock()
|
||||
defer m.lock.Unlock()
|
||||
|
||||
lease, exists := m.leases[id.NamespacedName()]
|
||||
if !exists {
|
||||
return nil, false
|
||||
}
|
||||
return lease, lease.Add(objectName)
|
||||
}
|
||||
|
||||
func (m *Manager) addLocked(ctx context.Context, id ID) *Lease {
|
||||
|
||||
// A lease whose context is already cancelled cannot be handed out again:
|
||||
// anything derived from it would be cancelled straight away. Replace it.
|
||||
if l, exists := m.leases[id.NamespacedName()]; !exists || l.Ctx.Err() != nil {
|
||||
leaseCtx, leaseCancel := context.WithCancel(ctx)
|
||||
m.leases[id.NamespacedName()] = newLease(leaseCtx, leaseCancel)
|
||||
}
|
||||
@@ -45,17 +68,53 @@ func (m *Manager) Add(ctx context.Context, id ID) *Lease {
|
||||
return m.leases[id.NamespacedName()]
|
||||
}
|
||||
|
||||
// Delete removes the lease and cancels it if the lease counter equals 0.
|
||||
func (m *Manager) Delete(id ID, objectName string) {
|
||||
// Delete removes the object from the lease it was added to and cancels that lease
|
||||
// once its last object is gone. It reports whether the lease was retired. With a
|
||||
// common lease, the siblings that still use it keep it alive.
|
||||
//
|
||||
// The lease the caller was given has to be passed in, because cleanup is usually
|
||||
// deferred to a goroutine that runs long after the object went away. By then the
|
||||
// lease of that name may already have been replaced, for instance because the
|
||||
// service was torn down and rebuilt, and cancelling the replacement would leave
|
||||
// the service unhandled. A stale caller is therefore ignored.
|
||||
//
|
||||
// Teardown paths have to call this synchronously rather than leaving it to the
|
||||
// deferred cleanup: until the lease is out of the map, Add hands the same
|
||||
// instance back, so a service that is rebuilt straight away gets parented to a
|
||||
// lease that the pending cleanup is about to cancel.
|
||||
func (m *Manager) Delete(id ID, objectName string, l *Lease) bool {
|
||||
m.lock.Lock()
|
||||
defer m.lock.Unlock()
|
||||
if _, exist := m.leases[id.NamespacedName()]; exist {
|
||||
m.leases[id.NamespacedName()].delete(objectName)
|
||||
if m.leases[id.NamespacedName()].cnt.Load() < 1 {
|
||||
m.leases[id.NamespacedName()].Cancel()
|
||||
delete(m.leases, id.NamespacedName())
|
||||
}
|
||||
|
||||
current := m.currentFor(id, l)
|
||||
if current == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
current.delete(objectName)
|
||||
if current.cnt.Load() < 1 {
|
||||
m.retire(id, current)
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// currentFor returns the registered lease for id, or nil when the caller is
|
||||
// stale, meaning the lease it holds is no longer the registered one. Callers have
|
||||
// to hold m.lock.
|
||||
func (m *Manager) currentFor(id ID, l *Lease) *Lease {
|
||||
current, exist := m.leases[id.NamespacedName()]
|
||||
if !exist || (l != nil && current != l) {
|
||||
return nil
|
||||
}
|
||||
return current
|
||||
}
|
||||
|
||||
// retire cancels the lease and drops it from the manager. Callers have to hold
|
||||
// m.lock.
|
||||
func (m *Manager) retire(id ID, l *Lease) {
|
||||
l.Cancel()
|
||||
delete(m.leases, id.NamespacedName())
|
||||
}
|
||||
|
||||
// Get returns lease for the service.
|
||||
@@ -73,62 +132,164 @@ func (m *Manager) Get(id ID) *Lease {
|
||||
type Lease struct {
|
||||
Ctx context.Context
|
||||
Cancel context.CancelFunc
|
||||
Started chan any
|
||||
services sync.Map
|
||||
cnt atomic.Int64
|
||||
Elected atomic.Bool
|
||||
Mtx sync.Mutex
|
||||
locked bool
|
||||
stateMu sync.Mutex
|
||||
running bool
|
||||
ended uint64
|
||||
changed chan struct{}
|
||||
}
|
||||
|
||||
func newLease(ctx context.Context, cancel context.CancelFunc) *Lease {
|
||||
return &Lease{
|
||||
Ctx: ctx,
|
||||
Cancel: cancel,
|
||||
Started: make(chan any),
|
||||
changed: make(chan struct{}),
|
||||
}
|
||||
}
|
||||
|
||||
// NewElectionContext returns a context for one election runner. Cancelling it
|
||||
// stops only that runner; the Lease context remains live until its final member
|
||||
// is deleted from the Manager.
|
||||
func (l *Lease) NewElectionContext(parent context.Context) (context.Context, context.CancelFunc) {
|
||||
ctx, cancel := context.WithCancel(l.Ctx)
|
||||
stopParent := context.AfterFunc(parent, cancel)
|
||||
return ctx, func() {
|
||||
stopParent()
|
||||
cancel()
|
||||
}
|
||||
}
|
||||
|
||||
// Add adds the object to the lease and increments counter
|
||||
// it will return true if object was added
|
||||
func (l *Lease) Add(name string) bool {
|
||||
if _, exists := l.services.Load(name); !exists {
|
||||
l.services.Store(name, true)
|
||||
if _, exists := l.services.LoadOrStore(name, true); !exists {
|
||||
l.cnt.Add(1)
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// delete removes the service from the lease and decrements the counter
|
||||
// delete removes the service from the lease and decrements the counter.
|
||||
func (l *Lease) delete(service string) {
|
||||
if _, exists := l.services.Load(service); exists {
|
||||
l.services.Delete(service)
|
||||
if _, exists := l.services.LoadAndDelete(service); exists {
|
||||
l.cnt.Add(-1)
|
||||
}
|
||||
}
|
||||
|
||||
func (l *Lease) Lock() {
|
||||
l.Mtx.Lock()
|
||||
l.locked = true
|
||||
func (l *Lease) BeginElection() bool {
|
||||
l.stateMu.Lock()
|
||||
defer l.stateMu.Unlock()
|
||||
if l.Elected.Load() || l.running {
|
||||
return false
|
||||
}
|
||||
l.running = true
|
||||
l.signalStateLocked()
|
||||
return true
|
||||
}
|
||||
|
||||
func (l *Lease) Unlock() {
|
||||
if l.locked {
|
||||
l.locked = false
|
||||
l.Mtx.Unlock()
|
||||
func (l *Lease) ElectionStarted() {
|
||||
l.stateMu.Lock()
|
||||
defer l.stateMu.Unlock()
|
||||
if l.Elected.Load() {
|
||||
return
|
||||
}
|
||||
l.Elected.Store(true)
|
||||
l.running = false
|
||||
l.signalStateLocked()
|
||||
}
|
||||
|
||||
func (l *Lease) ElectionStopped() {
|
||||
l.stateMu.Lock()
|
||||
defer l.stateMu.Unlock()
|
||||
if !l.Elected.Load() && !l.running {
|
||||
return
|
||||
}
|
||||
l.Elected.Store(false)
|
||||
l.running = false
|
||||
l.ended++
|
||||
l.signalStateLocked()
|
||||
}
|
||||
|
||||
// WaitForLeader waits for an in-flight lease election to either elect a leader
|
||||
// or finish without one. It never holds the lease state mutex while waiting.
|
||||
func (l *Lease) WaitForLeader(ctx context.Context) bool {
|
||||
_, elected := l.WaitForLeaderGeneration(ctx)
|
||||
return elected
|
||||
}
|
||||
|
||||
// WaitForLeaderGeneration waits for leadership and returns the election-end
|
||||
// generation observed atomically with the elected state.
|
||||
func (l *Lease) WaitForLeaderGeneration(ctx context.Context) (uint64, bool) {
|
||||
for {
|
||||
elected, running, changed, ended := l.state()
|
||||
if elected {
|
||||
return ended, true
|
||||
}
|
||||
if !running {
|
||||
return 0, false
|
||||
}
|
||||
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return 0, false
|
||||
case <-l.Ctx.Done():
|
||||
return 0, false
|
||||
case <-changed:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// WaitForElectionEnd waits until an elected lease loses its leader. It never
|
||||
// holds the lease state mutex while waiting.
|
||||
func (l *Lease) WaitForElectionEnd(ctx context.Context) {
|
||||
_, _, _, initialEnded := l.state()
|
||||
l.WaitForElectionEndAfter(ctx, initialEnded)
|
||||
}
|
||||
|
||||
// WaitForElectionEndAfter waits until the leadership generation returned by
|
||||
// WaitForLeaderGeneration ends, even if a replacement election starts first.
|
||||
func (l *Lease) WaitForElectionEndAfter(ctx context.Context, initialEnded uint64) {
|
||||
for {
|
||||
elected, _, changed, ended := l.state()
|
||||
if !elected || ended != initialEnded {
|
||||
return
|
||||
}
|
||||
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-l.Ctx.Done():
|
||||
return
|
||||
case <-changed:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (l *Lease) state() (bool, bool, <-chan struct{}, uint64) {
|
||||
l.stateMu.Lock()
|
||||
defer l.stateMu.Unlock()
|
||||
return l.Elected.Load(), l.running, l.changed, l.ended
|
||||
}
|
||||
|
||||
func (l *Lease) signalStateLocked() {
|
||||
close(l.changed)
|
||||
l.changed = make(chan struct{})
|
||||
}
|
||||
|
||||
// ServiceName gets lease name and id for the service.
|
||||
func ServiceName(service *v1.Service) (string, string) {
|
||||
name, exists := service.Annotations[kubevip.ServiceLease]
|
||||
if !exists || name == "" {
|
||||
name = fmt.Sprintf("kubevip-%s", service.Name)
|
||||
return ServiceNameFor(service.Namespace, service.Name, service.Annotations[kubevip.ServiceLease])
|
||||
}
|
||||
|
||||
func ServiceNameFor(namespace, serviceName, leaseName string) (string, string) {
|
||||
name := leaseName
|
||||
if name == "" {
|
||||
name = fmt.Sprintf("kubevip-%s", serviceName)
|
||||
}
|
||||
|
||||
serviceLeaseParts := strings.Split(name, "/")
|
||||
namespace := service.Namespace
|
||||
|
||||
if len(serviceLeaseParts) > 1 {
|
||||
namespace = serviceLeaseParts[0]
|
||||
|
||||
@@ -2,6 +2,7 @@ package lease
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -33,6 +34,391 @@ func getSvcData(svc *v1.Service) (context.Context, ID) {
|
||||
|
||||
const serviceLeaseAnnotation = kubevip.ServiceLease
|
||||
|
||||
func TestServiceNameForMatchesServiceName(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
namespace string
|
||||
service string
|
||||
lease string
|
||||
wantNamespace string
|
||||
wantName string
|
||||
}{
|
||||
{name: "default lease", namespace: "default", service: "api", wantNamespace: "default", wantName: "kubevip-api"},
|
||||
{name: "named lease", namespace: "default", service: "api", lease: "shared", wantNamespace: "default", wantName: "shared"},
|
||||
{name: "cross-namespace lease", namespace: "default", service: "api", lease: "leases/shared", wantNamespace: "leases", wantName: "shared"},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
service := createTestService(test.service, test.namespace, map[string]string{kubevip.ServiceLease: test.lease})
|
||||
for name, serviceName := range map[string]func() (string, string){
|
||||
"ServiceName": func() (string, string) { return ServiceName(service) },
|
||||
"ServiceNameFor": func() (string, string) { return ServiceNameFor(test.namespace, test.service, test.lease) },
|
||||
} {
|
||||
namespace, leaseName := serviceName()
|
||||
if namespace != test.wantNamespace || leaseName != test.wantName {
|
||||
t.Errorf("%s() = %s/%s, want %s/%s", name, namespace, leaseName, test.wantNamespace, test.wantName)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func electLease(t *testing.T, lease *Lease) {
|
||||
t.Helper()
|
||||
if !lease.BeginElection() {
|
||||
t.Fatal("expected lease to admit an election candidate")
|
||||
}
|
||||
lease.ElectionStarted()
|
||||
}
|
||||
|
||||
func TestManagerAcquireRegistersMembership(t *testing.T) {
|
||||
manager := NewManager()
|
||||
service := createTestService("service", "default", nil)
|
||||
id := getSvcID(service)
|
||||
objectName := ServiceNamespacedName(service)
|
||||
|
||||
lease, first := manager.Acquire(context.Background(), id, objectName)
|
||||
if !first {
|
||||
t.Fatal("first acquire did not register the service")
|
||||
}
|
||||
if _, second := manager.Acquire(context.Background(), id, objectName); second {
|
||||
t.Fatal("second acquire registered the same service twice")
|
||||
}
|
||||
|
||||
manager.Delete(id, objectName, lease)
|
||||
if manager.Get(id) != nil {
|
||||
t.Fatal("lease remained after its only acquired member was deleted")
|
||||
}
|
||||
}
|
||||
|
||||
func TestElectionContextCancellationDoesNotCancelSharedLease(t *testing.T) {
|
||||
manager := NewManager()
|
||||
id := NewID("kubernetes", "default", "shared")
|
||||
sharedLease, _ := manager.Acquire(context.Background(), id, "control-plane")
|
||||
if claimed, _ := manager.Claim(id, "service"); claimed != sharedLease {
|
||||
t.Fatal("second member did not join the shared lease")
|
||||
}
|
||||
|
||||
electionCtx, cancelElection := sharedLease.NewElectionContext(context.Background())
|
||||
cancelElection()
|
||||
select {
|
||||
case <-electionCtx.Done():
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("election context was not cancelled")
|
||||
}
|
||||
if sharedLease.Ctx.Err() != nil || manager.Get(id) != sharedLease {
|
||||
t.Fatal("cancelling one election runner cancelled the shared lease")
|
||||
}
|
||||
|
||||
memberCtx, cancelMember := sharedLease.NewElectionContext(context.Background())
|
||||
defer cancelMember()
|
||||
if manager.Delete(id, "control-plane", sharedLease) {
|
||||
t.Fatal("deleting one member retired a shared lease")
|
||||
}
|
||||
if !manager.Delete(id, "service", sharedLease) {
|
||||
t.Fatal("deleting the final member did not retire the lease")
|
||||
}
|
||||
select {
|
||||
case <-memberCtx.Done():
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("retiring the shared lease did not cancel an election context")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWaitForElectionEndObservesRapidRestart(t *testing.T) {
|
||||
manager := NewManager()
|
||||
service := createTestService("service", "default", nil)
|
||||
serviceLease := manager.Add(context.Background(), getSvcID(service))
|
||||
if !serviceLease.BeginElection() {
|
||||
t.Fatal("first election did not start")
|
||||
}
|
||||
serviceLease.ElectionStarted()
|
||||
|
||||
initialEnded, elected := serviceLease.WaitForLeaderGeneration(context.Background())
|
||||
if !elected {
|
||||
t.Fatal("waiter did not observe the elected lease")
|
||||
}
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
serviceLease.WaitForElectionEndAfter(context.Background(), initialEnded)
|
||||
close(done)
|
||||
}()
|
||||
serviceLease.ElectionStopped()
|
||||
if !serviceLease.BeginElection() {
|
||||
t.Fatal("replacement election did not start")
|
||||
}
|
||||
serviceLease.ElectionStarted()
|
||||
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("waiter missed election end during rapid restart")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerClaimDoesNotCreateRetiredLease(t *testing.T) {
|
||||
manager := NewManager()
|
||||
service := createTestService("service", "default", nil)
|
||||
if lease, joined := manager.Claim(getSvcID(service), ServiceNamespacedName(service)); lease != nil || joined {
|
||||
t.Fatal("claim created or joined a lease that does not exist")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLeaseElectionStateCoordinatesCandidates(t *testing.T) {
|
||||
leaseCtx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
lease := newLease(leaseCtx, cancel)
|
||||
|
||||
if !lease.BeginElection() {
|
||||
t.Fatal("first election candidate was not admitted")
|
||||
}
|
||||
if lease.BeginElection() {
|
||||
t.Fatal("second election candidate was admitted while election was running")
|
||||
}
|
||||
|
||||
joined := make(chan bool, 1)
|
||||
go func() {
|
||||
joined <- lease.WaitForLeader(context.Background())
|
||||
}()
|
||||
lease.ElectionStarted()
|
||||
select {
|
||||
case elected := <-joined:
|
||||
if !elected {
|
||||
t.Fatal("follower did not observe elected lease")
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("follower remained blocked after election succeeded")
|
||||
}
|
||||
|
||||
lease.ElectionStopped()
|
||||
if !lease.BeginElection() {
|
||||
t.Fatal("lease did not admit a new candidate after election stopped")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLeaseWaitForLeaderReturnsWhenCandidateStops(t *testing.T) {
|
||||
leaseCtx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
lease := newLease(leaseCtx, cancel)
|
||||
if !lease.BeginElection() {
|
||||
t.Fatal("candidate was not admitted")
|
||||
}
|
||||
|
||||
joined := make(chan bool, 1)
|
||||
go func() {
|
||||
joined <- lease.WaitForLeader(context.Background())
|
||||
}()
|
||||
lease.ElectionStopped()
|
||||
select {
|
||||
case elected := <-joined:
|
||||
if elected {
|
||||
t.Fatal("follower observed a leader after candidate stopped")
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("follower remained blocked after candidate stopped")
|
||||
}
|
||||
}
|
||||
|
||||
// TestLeaseSupportsRetakingElectionAfterCandidateStops exercises the retry
|
||||
// StartCluster relies on for a shared control-plane/Services lease: once
|
||||
// WaitForLeader reports the campaign ended without ever electing a leader, a
|
||||
// waiter must be able to begin its own election immediately instead of being
|
||||
// left with no active runner.
|
||||
func TestLeaseSupportsRetakingElectionAfterCandidateStops(t *testing.T) {
|
||||
leaseCtx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
lease := newLease(leaseCtx, cancel)
|
||||
if !lease.BeginElection() {
|
||||
t.Fatal("candidate was not admitted")
|
||||
}
|
||||
|
||||
retried := make(chan bool, 1)
|
||||
go func() {
|
||||
if lease.WaitForLeader(context.Background()) {
|
||||
retried <- false
|
||||
return
|
||||
}
|
||||
retried <- lease.BeginElection()
|
||||
}()
|
||||
lease.ElectionStopped()
|
||||
|
||||
select {
|
||||
case tookOver := <-retried:
|
||||
if !tookOver {
|
||||
t.Fatal("waiter could not begin its own election after the shared campaign ended without a leader")
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("waiter remained blocked after candidate stopped")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLeaseWaitForElectionEndReleasesFollowers(t *testing.T) {
|
||||
leaseCtx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
lease := newLease(leaseCtx, cancel)
|
||||
if !lease.BeginElection() {
|
||||
t.Fatal("candidate was not admitted")
|
||||
}
|
||||
lease.ElectionStarted()
|
||||
|
||||
finished := make(chan struct{})
|
||||
go func() {
|
||||
lease.WaitForElectionEnd(context.Background())
|
||||
close(finished)
|
||||
}()
|
||||
lease.ElectionStopped()
|
||||
select {
|
||||
case <-finished:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("follower remained blocked after leadership stopped")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLeaseWaitForLeaderReturnsWhenContextCancelled(t *testing.T) {
|
||||
for _, cancelWait := range []struct {
|
||||
name string
|
||||
cancel func(context.CancelFunc, context.CancelFunc)
|
||||
}{
|
||||
{"caller context", func(cancelCaller, _ context.CancelFunc) { cancelCaller() }},
|
||||
{"lease context", func(_, cancelLease context.CancelFunc) { cancelLease() }},
|
||||
} {
|
||||
t.Run(cancelWait.name, func(t *testing.T) {
|
||||
leaseCtx, cancelLease := context.WithCancel(context.Background())
|
||||
defer cancelLease()
|
||||
lease := newLease(leaseCtx, cancelLease)
|
||||
if !lease.BeginElection() {
|
||||
t.Fatal("candidate was not admitted")
|
||||
}
|
||||
|
||||
callerCtx, cancelCaller := context.WithCancel(context.Background())
|
||||
defer cancelCaller()
|
||||
result := make(chan bool, 1)
|
||||
go func() {
|
||||
result <- lease.WaitForLeader(callerCtx)
|
||||
}()
|
||||
|
||||
cancelWait.cancel(cancelCaller, cancelLease)
|
||||
select {
|
||||
case elected := <-result:
|
||||
if elected {
|
||||
t.Fatal("waiter observed a leader after cancellation")
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("waiter remained blocked after cancellation")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestLeaseWaitForElectionEndReturnsWhenContextCancelled(t *testing.T) {
|
||||
for _, cancelWait := range []struct {
|
||||
name string
|
||||
cancel func(context.CancelFunc, context.CancelFunc)
|
||||
}{
|
||||
{"caller context", func(cancelCaller, _ context.CancelFunc) { cancelCaller() }},
|
||||
{"lease context", func(_, cancelLease context.CancelFunc) { cancelLease() }},
|
||||
} {
|
||||
t.Run(cancelWait.name, func(t *testing.T) {
|
||||
leaseCtx, cancelLease := context.WithCancel(context.Background())
|
||||
defer cancelLease()
|
||||
lease := newLease(leaseCtx, cancelLease)
|
||||
electLease(t, lease)
|
||||
|
||||
callerCtx, cancelCaller := context.WithCancel(context.Background())
|
||||
defer cancelCaller()
|
||||
finished := make(chan struct{})
|
||||
go func() {
|
||||
lease.WaitForElectionEnd(callerCtx)
|
||||
close(finished)
|
||||
}()
|
||||
|
||||
cancelWait.cancel(cancelCaller, cancelLease)
|
||||
select {
|
||||
case <-finished:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("waiter remained blocked after cancellation")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerAcquireRegistersConcurrentMemberOnce(t *testing.T) {
|
||||
manager := NewManager()
|
||||
service := createTestService("service", "default", nil)
|
||||
id := getSvcID(service)
|
||||
objectName := ServiceNamespacedName(service)
|
||||
|
||||
type result struct {
|
||||
lease *Lease
|
||||
isNew bool
|
||||
}
|
||||
results := make(chan result, 64)
|
||||
var wg sync.WaitGroup
|
||||
for range cap(results) {
|
||||
wg.Go(func() {
|
||||
lease, isNew := manager.Acquire(context.Background(), id, objectName)
|
||||
results <- result{lease, isNew}
|
||||
})
|
||||
}
|
||||
wg.Wait()
|
||||
close(results)
|
||||
|
||||
var lease *Lease
|
||||
newMembers := 0
|
||||
for result := range results {
|
||||
if lease == nil {
|
||||
lease = result.lease
|
||||
} else if result.lease != lease {
|
||||
t.Fatal("concurrent acquires returned different leases")
|
||||
}
|
||||
if result.isNew {
|
||||
newMembers++
|
||||
}
|
||||
}
|
||||
if newMembers != 1 {
|
||||
t.Fatalf("new member registrations = %d, want 1", newMembers)
|
||||
}
|
||||
|
||||
manager.Delete(id, objectName, lease)
|
||||
}
|
||||
|
||||
func TestManagerClaimRegistersConcurrentMemberOnce(t *testing.T) {
|
||||
manager := NewManager()
|
||||
service := createTestService("service", "default", nil)
|
||||
id := getSvcID(service)
|
||||
objectName := ServiceNamespacedName(service)
|
||||
lease := manager.Add(context.Background(), id)
|
||||
|
||||
type result struct {
|
||||
lease *Lease
|
||||
isNew bool
|
||||
}
|
||||
results := make(chan result, 64)
|
||||
var wg sync.WaitGroup
|
||||
for range cap(results) {
|
||||
wg.Go(func() {
|
||||
claimed, isNew := manager.Claim(id, objectName)
|
||||
results <- result{claimed, isNew}
|
||||
})
|
||||
}
|
||||
wg.Wait()
|
||||
close(results)
|
||||
|
||||
newMembers := 0
|
||||
for result := range results {
|
||||
if result.lease != lease {
|
||||
t.Fatal("concurrent claims returned a different lease")
|
||||
}
|
||||
if result.isNew {
|
||||
newMembers++
|
||||
}
|
||||
}
|
||||
if newMembers != 1 {
|
||||
t.Fatalf("new member registrations = %d, want 1", newMembers)
|
||||
}
|
||||
|
||||
manager.Delete(id, objectName, lease)
|
||||
}
|
||||
|
||||
// TestManager_Add_NewLease tests adding a new service with a new lease
|
||||
func TestManager_Add_NewLease(t *testing.T) {
|
||||
mgr := NewManager()
|
||||
@@ -53,10 +439,6 @@ func TestManager_Add_NewLease(t *testing.T) {
|
||||
if leaseID.Cancel == nil {
|
||||
t.Error("expected lease cancel func to be non-nil")
|
||||
}
|
||||
if leaseID.Started == nil {
|
||||
t.Error("expected lease Started channel to be non-nil")
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
// TestManager_Add_ExistingLease tests adding a service with an existing lease
|
||||
@@ -99,7 +481,7 @@ func TestManager_Delete_DecrementCounter(t *testing.T) {
|
||||
|
||||
// Delete once - should remove the lease
|
||||
|
||||
mgr.Delete(leaseID1, objectName)
|
||||
mgr.Delete(leaseID1, objectName, nil)
|
||||
|
||||
lease := mgr.Get(getSvcID(svc))
|
||||
if lease != nil {
|
||||
@@ -127,7 +509,7 @@ func TestManager_Delete_CancelsContext(t *testing.T) {
|
||||
}
|
||||
|
||||
// Delete the lease
|
||||
mgr.Delete(leaseID1, objectName)
|
||||
mgr.Delete(leaseID1, objectName, nil)
|
||||
|
||||
// Verify context is cancelled
|
||||
select {
|
||||
@@ -149,7 +531,7 @@ func TestManager_Add_AfterDelete_CreatesNewLease(t *testing.T) {
|
||||
lease1 := mgr.Add(ctx1, leaseID1)
|
||||
_ = lease1.Add(objectName)
|
||||
|
||||
mgr.Delete(leaseID1, objectName)
|
||||
mgr.Delete(leaseID1, objectName, nil)
|
||||
|
||||
ctx2, leaseID2 := getSvcData(svc)
|
||||
lease2 := mgr.Add(ctx2, leaseID2)
|
||||
@@ -244,7 +626,7 @@ func TestManager_ConcurrentAccess(t *testing.T) {
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
mgr.Delete(leaseID1, objectName1)
|
||||
mgr.Delete(leaseID1, objectName1, nil)
|
||||
|
||||
// After a one delete, lease should be gone
|
||||
lease := mgr.Get(getSvcID(svc))
|
||||
@@ -253,33 +635,6 @@ func TestManager_ConcurrentAccess(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestLease_StartedChannel tests the Started channel behavior
|
||||
func TestLease_StartedChannel(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
lease := newLease(ctx, cancel)
|
||||
|
||||
// Started channel should be open initially
|
||||
select {
|
||||
case <-lease.Started:
|
||||
t.Fatal("expected Started channel to be open initially")
|
||||
default:
|
||||
// Expected
|
||||
}
|
||||
|
||||
// Close the channel
|
||||
close(lease.Started)
|
||||
|
||||
// Now it should be closed
|
||||
select {
|
||||
case <-lease.Started:
|
||||
// Expected
|
||||
default:
|
||||
t.Error("expected Started channel to be closed after close()")
|
||||
}
|
||||
}
|
||||
|
||||
// TestGetName_WithoutAnnotation tests with no annotation
|
||||
func TestGetName_WithoutAnnotation(t *testing.T) {
|
||||
svc := createTestService("my-service", "my-namespace", nil)
|
||||
@@ -422,12 +777,12 @@ func TestManager_LeaderElectionRestartScenario_etcd(t *testing.T) {
|
||||
t.Fatal("expected first add to return isNew=true")
|
||||
}
|
||||
|
||||
// Simulate leadership acquired - close Started channel
|
||||
close(lease1.Started)
|
||||
// Simulate leadership acquired.
|
||||
electLease(t, lease1)
|
||||
|
||||
// Simulate leadership lost - the leader election function should delete the lease
|
||||
// This is the fix: delete the lease when RunOrDie returns
|
||||
mgr.Delete(leaseID1, objectName1)
|
||||
mgr.Delete(leaseID1, objectName1, nil)
|
||||
|
||||
// Verify lease is removed
|
||||
if mgr.Get(getSvcID(svc)) != nil {
|
||||
@@ -437,22 +792,17 @@ func TestManager_LeaderElectionRestartScenario_etcd(t *testing.T) {
|
||||
// Simulate restartable service watcher calling StartServicesLeaderElection again
|
||||
ctx2, leaseID2 := getSvcData(svc)
|
||||
lease2 := mgr.Add(ctx2, leaseID2)
|
||||
isNew2 := lease1.Add(objectName1)
|
||||
isNew2 := lease2.Add(objectName1)
|
||||
if !isNew2 {
|
||||
t.Fatal("expected second add after delete to return isNew=true")
|
||||
}
|
||||
|
||||
// Verify we got a new lease with a fresh Started channel
|
||||
// Verify we got a new lease with no elected leader.
|
||||
if lease1 == lease2 {
|
||||
t.Error("expected new lease to be different from old lease")
|
||||
}
|
||||
|
||||
// Verify the new Started channel is not closed
|
||||
select {
|
||||
case <-lease2.Started:
|
||||
t.Error("expected new lease's Started channel to be open")
|
||||
default:
|
||||
// Expected
|
||||
if lease2.Elected.Load() {
|
||||
t.Error("expected new lease to have no elected leader")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -478,8 +828,11 @@ func TestManager_CommonLeaseScenario(t *testing.T) {
|
||||
t.Error("expected first add to return isNew=true")
|
||||
}
|
||||
|
||||
// Simulate first service starting leadership
|
||||
close(lease1.Started)
|
||||
// Simulate first service starting leadership.
|
||||
if !lease1.BeginElection() {
|
||||
t.Fatal("expected first service to become election candidate")
|
||||
}
|
||||
lease1.ElectionStarted()
|
||||
|
||||
objectName2 := ServiceNamespacedName(svc2)
|
||||
|
||||
@@ -494,15 +847,18 @@ func TestManager_CommonLeaseScenario(t *testing.T) {
|
||||
if lease1 != lease2 {
|
||||
t.Error("expected same lease for services with same lease annotation")
|
||||
}
|
||||
if !lease2.WaitForLeader(context.Background()) {
|
||||
t.Fatal("shared-lease follower did not observe the elected lease")
|
||||
}
|
||||
|
||||
// Delete first service - lease should still exist
|
||||
mgr.Delete(leaseID1, objectName1)
|
||||
mgr.Delete(leaseID1, objectName1, nil)
|
||||
if mgr.Get(getSvcID(svc1)) == nil {
|
||||
t.Error("expected lease to still exist after first delete")
|
||||
}
|
||||
|
||||
// Delete second service - lease should be removed
|
||||
mgr.Delete(leaseID2, objectName2)
|
||||
mgr.Delete(leaseID2, objectName2, nil)
|
||||
if mgr.Get(getSvcID(svc2)) != nil {
|
||||
t.Error("expected lease to be removed after all services deleted")
|
||||
}
|
||||
@@ -525,8 +881,8 @@ func TestManager_RaceCondition_LeaseExistsBeforeDelete(t *testing.T) {
|
||||
t.Fatal("expected first add to return isNew=true")
|
||||
}
|
||||
|
||||
// Simulate leadership acquired - close Started channel
|
||||
close(lease1.Started)
|
||||
// Simulate leadership acquired.
|
||||
electLease(t, lease1)
|
||||
|
||||
// Simulate a second goroutine calling Add BEFORE the first goroutine's defer deletes the lease
|
||||
// This is the race condition scenario
|
||||
@@ -541,16 +897,12 @@ func TestManager_RaceCondition_LeaseExistsBeforeDelete(t *testing.T) {
|
||||
t.Error("expected same lease to be returned")
|
||||
}
|
||||
|
||||
// The Started channel should be closed (from the first run)
|
||||
select {
|
||||
case <-lease2.Started:
|
||||
// Expected - channel is closed
|
||||
default:
|
||||
t.Error("expected Started channel to be closed")
|
||||
if !lease2.Elected.Load() {
|
||||
t.Error("expected lease to remain elected")
|
||||
}
|
||||
|
||||
// Now the first goroutine's defer deletes the lease
|
||||
mgr.Delete(leaseID1, objectName1)
|
||||
mgr.Delete(leaseID1, objectName1, nil)
|
||||
|
||||
// The lease should not still exist because same service was processed twice, so we do not increment the counter
|
||||
if mgr.Get(getSvcID(svc)) != nil {
|
||||
@@ -558,7 +910,7 @@ func TestManager_RaceCondition_LeaseExistsBeforeDelete(t *testing.T) {
|
||||
}
|
||||
|
||||
// Second delete does nothing
|
||||
mgr.Delete(leaseID2, objectName1)
|
||||
mgr.Delete(leaseID2, objectName1, nil)
|
||||
if mgr.Get(getSvcID(svc)) != nil {
|
||||
t.Error("expected lease to not exist")
|
||||
}
|
||||
@@ -580,8 +932,8 @@ func TestManager_NonCommonLease_MultipleAdds(t *testing.T) {
|
||||
t.Error("expected first add to return isNew=true")
|
||||
}
|
||||
|
||||
// Close Started to simulate leadership acquired
|
||||
close(lease1.Started)
|
||||
// Simulate leadership acquired.
|
||||
electLease(t, lease1)
|
||||
|
||||
// Second Add (simulating another goroutine or restart attempt)
|
||||
ctx2, leaseID2 := getSvcData(svc)
|
||||
@@ -606,17 +958,17 @@ func TestManager_NonCommonLease_MultipleAdds(t *testing.T) {
|
||||
}
|
||||
|
||||
// Need one delete to remove the lease, another delete runs do nothing
|
||||
mgr.Delete(leaseID1, objectName1)
|
||||
mgr.Delete(leaseID1, objectName1, nil)
|
||||
if mgr.Get(getSvcID(svc)) != nil {
|
||||
t.Error("expected lease to be deleted")
|
||||
}
|
||||
|
||||
mgr.Delete(leaseID2, objectName1)
|
||||
mgr.Delete(leaseID2, objectName1, nil)
|
||||
if mgr.Get(getSvcID(svc)) != nil {
|
||||
t.Error("expected lease to be deleted")
|
||||
}
|
||||
|
||||
mgr.Delete(leaseID3, objectName1)
|
||||
mgr.Delete(leaseID3, objectName1, nil)
|
||||
if mgr.Get(getSvcID(svc)) != nil {
|
||||
t.Error("expected lease to be deleted")
|
||||
}
|
||||
@@ -648,12 +1000,8 @@ func TestManager_LeaseContextCancelledBeforeStarted(t *testing.T) {
|
||||
t.Error("expected second add to return isNew=false")
|
||||
}
|
||||
|
||||
// Verify Started is not closed yet
|
||||
select {
|
||||
case <-lease2.Started:
|
||||
t.Error("expected Started channel to be open")
|
||||
default:
|
||||
// Expected
|
||||
if lease2.Elected.Load() {
|
||||
t.Error("expected lease to have no elected leader")
|
||||
}
|
||||
|
||||
// Cancel the lease context (simulating timeout or leadership loss before acquiring)
|
||||
@@ -668,8 +1016,8 @@ func TestManager_LeaseContextCancelledBeforeStarted(t *testing.T) {
|
||||
}
|
||||
|
||||
// Delete should still work
|
||||
mgr.Delete(leaseID1, objectName1)
|
||||
mgr.Delete(leaseID2, objectName1)
|
||||
mgr.Delete(leaseID1, objectName1, nil)
|
||||
mgr.Delete(leaseID2, objectName1, nil)
|
||||
|
||||
if mgr.Get(getSvcID(svc)) != nil {
|
||||
t.Error("expected lease to be removed")
|
||||
@@ -693,7 +1041,7 @@ func TestManager_RestartAfterLeaseContextCancelled(t *testing.T) {
|
||||
lease1.Cancel()
|
||||
|
||||
// Delete the lease
|
||||
mgr.Delete(leaseID1, objectName1)
|
||||
mgr.Delete(leaseID1, objectName1, nil)
|
||||
|
||||
// Verify lease is gone
|
||||
if mgr.Get(getSvcID(svc)) != nil {
|
||||
@@ -709,7 +1057,7 @@ func TestManager_RestartAfterLeaseContextCancelled(t *testing.T) {
|
||||
t.Error("expected new lease after delete")
|
||||
}
|
||||
|
||||
// Verify new lease has fresh context and Started channel
|
||||
// Verify new lease has a fresh active context and no elected leader.
|
||||
select {
|
||||
case <-lease2.Ctx.Done():
|
||||
t.Error("expected new lease context to be active")
|
||||
@@ -717,11 +1065,8 @@ func TestManager_RestartAfterLeaseContextCancelled(t *testing.T) {
|
||||
// Expected
|
||||
}
|
||||
|
||||
select {
|
||||
case <-lease2.Started:
|
||||
t.Error("expected new lease Started channel to be open")
|
||||
default:
|
||||
// Expected
|
||||
if lease2.Elected.Load() {
|
||||
t.Error("expected new lease to have no elected leader")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -744,8 +1089,8 @@ func TestManager_NonCommonLease_WaitForLeaseContextDone(t *testing.T) {
|
||||
t.Fatal("expected first add to return isNew=true")
|
||||
}
|
||||
|
||||
// Simulate leadership acquired
|
||||
close(lease1.Started)
|
||||
// Simulate leadership acquired.
|
||||
electLease(t, lease1)
|
||||
|
||||
// Second Add - simulates another goroutine trying to start leader election
|
||||
// This should return isNew=false
|
||||
@@ -761,12 +1106,8 @@ func TestManager_NonCommonLease_WaitForLeaseContextDone(t *testing.T) {
|
||||
t.Error("expected same lease to be returned")
|
||||
}
|
||||
|
||||
// Verify Started channel is closed (leadership was acquired by first)
|
||||
select {
|
||||
case <-lease2.Started:
|
||||
// Expected - channel is closed
|
||||
default:
|
||||
t.Error("expected Started channel to be closed")
|
||||
if !lease2.Elected.Load() {
|
||||
t.Error("expected lease to remain elected")
|
||||
}
|
||||
|
||||
// In the actual code (leader.go), when isNew=false for non-common lease,
|
||||
@@ -794,11 +1135,11 @@ func TestManager_NonCommonLease_WaitForLeaseContextDone(t *testing.T) {
|
||||
}
|
||||
|
||||
// Now simulate the first leader election ending (defer deletes the lease)
|
||||
mgr.Delete(leaseID1, objectName1)
|
||||
mgr.Delete(leaseID1, objectName1, nil)
|
||||
|
||||
// The lease context should now be cancelled (because counter went to 0)
|
||||
// But we added twice, so we need to delete twice
|
||||
mgr.Delete(leaseID2, objectName1)
|
||||
mgr.Delete(leaseID2, objectName1, nil)
|
||||
|
||||
// Now the goroutine should have completed
|
||||
select {
|
||||
@@ -833,7 +1174,7 @@ func TestManager_NonCommonLease_SpinLoopPrevention(t *testing.T) {
|
||||
t.Fatal("expected first add to return isNew=true")
|
||||
}
|
||||
|
||||
close(lease1.Started)
|
||||
electLease(t, lease1)
|
||||
|
||||
// Track how many times Add is called in a tight loop
|
||||
// In the buggy code, this would spin forever
|
||||
@@ -878,7 +1219,7 @@ func TestManager_NonCommonLease_SpinLoopPrevention(t *testing.T) {
|
||||
t.Errorf("expected 100 adds, got %d", addCount)
|
||||
}
|
||||
|
||||
mgr.Delete(leaseID1, objectName1)
|
||||
mgr.Delete(leaseID1, objectName1, nil)
|
||||
if mgr.Get(getSvcID(svc)) != nil {
|
||||
t.Error("expected lease to be removed after first delete")
|
||||
}
|
||||
@@ -897,7 +1238,7 @@ func TestManager_NonCommonLease_ServiceContextCancellation(t *testing.T) {
|
||||
ctx1, leaseID1 := getSvcData(svc)
|
||||
lease1 := mgr.Add(ctx1, leaseID1)
|
||||
_ = lease1.Add(objectName1)
|
||||
close(lease1.Started)
|
||||
electLease(t, lease1)
|
||||
|
||||
// Second Add - returns isNew=false
|
||||
ctx2, leaseID2 := getSvcData(svc)
|
||||
@@ -937,3 +1278,179 @@ func TestManager_NonCommonLease_ServiceContextCancellation(t *testing.T) {
|
||||
t.Error("goroutine should have unblocked")
|
||||
}
|
||||
}
|
||||
|
||||
// TestManager_Delete_DoesNotCancelRecreatedLease reproduces the stale-cleanup bug.
|
||||
//
|
||||
// Every service that starts leader election also starts a goroutine that calls
|
||||
// Manager.Delete once the service context is cancelled. When a service is torn
|
||||
// down and immediately rebuilt, for instance because its externalTrafficPolicy
|
||||
// changed, that goroutine runs after the replacement lease was already created.
|
||||
// Deleting by name alone then cancels the live replacement, and the service is
|
||||
// never handled again.
|
||||
func TestManager_Delete_DoesNotCancelRecreatedLease(t *testing.T) {
|
||||
mgr := NewManager()
|
||||
svc := createTestService("test-svc", "default", nil)
|
||||
ctx, id := getSvcData(svc)
|
||||
objectName := ServiceNamespacedName(svc)
|
||||
|
||||
// The service is set up, and its lease is registered.
|
||||
old := mgr.Add(ctx, id)
|
||||
old.Add(objectName)
|
||||
|
||||
// The service is torn down and rebuilt straight away, so a fresh lease for the
|
||||
// same name exists before the old cleanup goroutine gets to run.
|
||||
mgr.Delete(id, objectName, old)
|
||||
fresh := mgr.Add(ctx, id)
|
||||
fresh.Add(objectName)
|
||||
|
||||
if old == fresh {
|
||||
t.Fatal("expected a new lease instance after delete")
|
||||
}
|
||||
|
||||
// Now the cleanup for the *old* lease finally runs. It has to be a no-op.
|
||||
mgr.Delete(id, objectName, old)
|
||||
|
||||
if fresh.Ctx.Err() != nil {
|
||||
t.Error("cleanup for the torn down lease cancelled the recreated lease")
|
||||
}
|
||||
if got := mgr.Get(id); got == nil {
|
||||
t.Error("cleanup for the torn down lease removed the recreated lease")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerDeleteDoesNotCancelReplacementAfterDirectLeaseCancellation(t *testing.T) {
|
||||
manager := NewManager()
|
||||
service := createTestService("service", "default", nil)
|
||||
id := getSvcID(service)
|
||||
objectName := ServiceNamespacedName(service)
|
||||
|
||||
old, isNew := manager.Acquire(context.Background(), id, objectName)
|
||||
if !isNew {
|
||||
t.Fatal("initial acquire did not register the service")
|
||||
}
|
||||
old.Cancel()
|
||||
|
||||
fresh, isNew := manager.Acquire(context.Background(), id, objectName)
|
||||
if !isNew {
|
||||
t.Fatal("replacement acquire did not register the service")
|
||||
}
|
||||
if fresh == old {
|
||||
t.Fatal("acquire reused a directly cancelled lease")
|
||||
}
|
||||
|
||||
manager.Delete(id, objectName, old)
|
||||
if fresh.Ctx.Err() != nil {
|
||||
t.Fatal("late cleanup for a directly cancelled lease cancelled its replacement")
|
||||
}
|
||||
if manager.Get(id) != fresh {
|
||||
t.Fatal("late cleanup for a directly cancelled lease removed its replacement")
|
||||
}
|
||||
|
||||
manager.Delete(id, objectName, fresh)
|
||||
}
|
||||
|
||||
// TestManager_Add_AfterCancelWithoutDelete_ReusesDoomedLease reproduces the
|
||||
// second half of the service rebuild race.
|
||||
//
|
||||
// A teardown cancels the service context but leaves the lease in the manager,
|
||||
// because the cleanup that removes it is deferred to a goroutine. If the
|
||||
// replacement service context is built before that goroutine runs, Add hands
|
||||
// back the very same lease instance, so the replacement is parented to a lease
|
||||
// that is about to be cancelled. The instance guard in Delete cannot help,
|
||||
// because the doomed lease and the current lease are the same object.
|
||||
//
|
||||
// Retiring the lease synchronously during teardown is what makes Add return a
|
||||
// genuinely fresh instance.
|
||||
func TestManager_Add_AfterCancelWithoutDelete_ReusesDoomedLease(t *testing.T) {
|
||||
mgr := NewManager()
|
||||
svc := createTestService("test-svc", "default", nil)
|
||||
ctx, id := getSvcData(svc)
|
||||
objectName := ServiceNamespacedName(svc)
|
||||
|
||||
old := mgr.Add(ctx, id)
|
||||
old.Add(objectName)
|
||||
|
||||
// Teardown drops the service from its lease synchronously, so the rebuild that
|
||||
// follows cannot be parented to it even though the deferred cleanup has not run.
|
||||
mgr.Delete(id, objectName, old)
|
||||
|
||||
fresh := mgr.Add(ctx, id)
|
||||
fresh.Add(objectName)
|
||||
|
||||
if fresh == old {
|
||||
t.Fatal("replacement service context would be parented to the doomed lease")
|
||||
}
|
||||
|
||||
// The deferred cleanup for the old lease now runs and must be a no-op.
|
||||
mgr.Delete(id, objectName, old)
|
||||
|
||||
if fresh.Ctx.Err() != nil {
|
||||
t.Error("late cleanup cancelled the replacement lease")
|
||||
}
|
||||
}
|
||||
|
||||
// TestManager_LeaseLifetimeInvariant pins the lifetime rule for the whole
|
||||
// Add/Delete surface rather than one scenario: a lease stays usable for exactly as
|
||||
// long as at least one object still holds it, and is replaced afterwards.
|
||||
//
|
||||
// That is the property the common lease depends on, and the one a per-lease
|
||||
// teardown breaks: dropping one service must not cancel a lease its siblings are
|
||||
// still using. Raised by Patryk in review of #1669.
|
||||
func TestManager_LeaseLifetimeInvariant(t *testing.T) {
|
||||
shared := map[string]string{serviceLeaseAnnotation: "shared-lease"}
|
||||
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
objects int
|
||||
}{
|
||||
{"single object", 1},
|
||||
{"two objects sharing a lease", 2},
|
||||
{"several objects sharing a lease", 4},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
mgr := NewManager()
|
||||
ctx, id := getSvcData(createTestService("svc0", "default", shared))
|
||||
|
||||
objects := make([]string, tc.objects)
|
||||
for i := range objects {
|
||||
objects[i] = ServiceNamespacedName(createTestService(fmt.Sprintf("svc%d", i), "default", shared))
|
||||
}
|
||||
|
||||
l := mgr.Add(ctx, id)
|
||||
for _, o := range objects {
|
||||
if !l.Add(o) {
|
||||
t.Fatalf("object %q was not added", o)
|
||||
}
|
||||
}
|
||||
|
||||
// Drop the objects one at a time. Every drop but the last has to leave
|
||||
// the lease usable, because the rest still depend on it.
|
||||
for i, o := range objects {
|
||||
mgr.Delete(id, o, l)
|
||||
|
||||
if remaining := len(objects) - i - 1; remaining > 0 {
|
||||
if l.Ctx.Err() != nil {
|
||||
t.Fatalf("lease was cancelled with %d object(s) still holding it", remaining)
|
||||
}
|
||||
if mgr.Get(id) != l {
|
||||
t.Fatalf("lease was dropped with %d object(s) still holding it", remaining)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
if l.Ctx.Err() == nil {
|
||||
t.Error("lease was not cancelled after its last object went away")
|
||||
}
|
||||
if mgr.Get(id) != nil {
|
||||
t.Error("lease was not removed after its last object went away")
|
||||
}
|
||||
}
|
||||
|
||||
// A rebuild has to get a genuinely fresh lease, so nothing derived from it
|
||||
// is cancelled by the teardown that just happened.
|
||||
if fresh := mgr.Add(ctx, id); fresh == l || fresh.Ctx.Err() != nil {
|
||||
t.Error("rebuild reused the retired lease")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -58,8 +58,9 @@ type IPVSLoadBalancer struct {
|
||||
|
||||
func NewIPVSLB(ctx context.Context, network vip.Network, port uint16, forwardingMethod string, backendHealthCheckInterval int,
|
||||
nftables bool, killFunc func(), wg *sync.WaitGroup) (*IPVSLoadBalancer, error) {
|
||||
log.Info("Starting IPVS LoadBalancer", "network", network)
|
||||
|
||||
address := network.IP()
|
||||
log.Info("Starting IPVS LoadBalancer", "address", network)
|
||||
|
||||
// Create IPVS client
|
||||
c, err := ipvs.New()
|
||||
@@ -239,7 +240,7 @@ func (lb *IPVSLoadBalancer) addBackend(address string, port uint16) error {
|
||||
// Fatal error at this point as IPVS is probably not working
|
||||
log.Error("Unable to create an IPVS service, ensure IPVS kernel modules are loaded")
|
||||
log.Error("IPVS service", "err", err)
|
||||
return utils.NewPanicError(fmt.Sprintf("unable to create an IPVS service - %s", err))
|
||||
return utils.WrapPanicError(err, "unable to create an IPVS service")
|
||||
|
||||
}
|
||||
log.Info("load-Balancer services created", "address", lb.addrString(), "port", lb.Port)
|
||||
|
||||
@@ -2,7 +2,6 @@ package manager
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"os"
|
||||
@@ -118,7 +117,8 @@ func New(ctx context.Context, configMap string, config *kubevip.Config) (*Manage
|
||||
switch {
|
||||
case config.LeaderElectionType == "etcd":
|
||||
// Do nothing, we don't construct a k8s client for etcd leader election
|
||||
case config.K8sConfigFile != "" && utils.FileExists(config.K8sConfigFile):
|
||||
case config.K8sConfigFile != "" && config.K8sConfigFile != adminConfigPath &&
|
||||
config.K8sConfigFile != homeConfigPath && utils.FileExists(config.K8sConfigFile):
|
||||
// An explicitly configured kubeconfig (k8s_config_file env or
|
||||
// --k8sConfigPath) takes precedence over the well-known host paths.
|
||||
// KubernetesAddr, when set, overrides the API endpoint - static pods
|
||||
@@ -413,7 +413,7 @@ func (sm *Manager) startMode(ctx context.Context) error {
|
||||
return nil
|
||||
default:
|
||||
if err = w.StartServices(modeCtx); err != nil {
|
||||
if errors.Is(err, &utils.PanicError{}) {
|
||||
if utils.IsPanicError(err) {
|
||||
sm.Kill()
|
||||
return fmt.Errorf("failed to reconcile services, non-recoverable error: %w", err)
|
||||
} else {
|
||||
|
||||
@@ -11,10 +11,9 @@ import (
|
||||
log "log/slog"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/utils"
|
||||
|
||||
"github.com/davecgh/go-spew/spew"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/labels"
|
||||
"k8s.io/client-go/kubernetes"
|
||||
@@ -42,6 +41,9 @@ func annotationsWatcher(ctx context.Context, clientSet,
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(nodeList.Items) == 0 {
|
||||
return fmt.Errorf("no node found with hostname %q", config.NodeName)
|
||||
}
|
||||
|
||||
// We'll assume there's only one node with the hostname annotation. If that's not true,
|
||||
// there's probably bigger problems
|
||||
@@ -108,22 +110,15 @@ func annotationsWatcher(ctx context.Context, clientSet,
|
||||
// Un-used
|
||||
case watch.Error:
|
||||
log.Error("Error attempting to watch Kubernetes Nodes")
|
||||
|
||||
// This round trip allows us to handle unstructured status
|
||||
errObject := apierrors.FromObject(event.Object)
|
||||
statusErr, ok := errObject.(*apierrors.StatusError)
|
||||
if !ok {
|
||||
log.Error(spew.Sprintf("Received an error which is not *metav1.Status but %#+v", event.Object))
|
||||
|
||||
}
|
||||
|
||||
status := statusErr.ErrStatus
|
||||
log.Error(status.String())
|
||||
log.Error("annotations watcher failed", "err", utils.WatchError(event.Object))
|
||||
default:
|
||||
}
|
||||
}
|
||||
log.Info("[annotations] exiting annotations watcher")
|
||||
return nil
|
||||
if ctx.Err() != nil {
|
||||
return nil
|
||||
}
|
||||
return utils.NewPanicError("annotations watcher channel closed unexpectedly")
|
||||
}
|
||||
|
||||
// parseNodeAnnotations parses the annotations on the node and updates the configuration
|
||||
|
||||
21
pkg/manager/watch_annotations_test.go
Normal file
21
pkg/manager/watch_annotations_test.go
Normal file
@@ -0,0 +1,21 @@
|
||||
package manager
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"k8s.io/client-go/kubernetes/fake"
|
||||
)
|
||||
|
||||
func TestAnnotationsWatcherHandlesEmptyNodeList(t *testing.T) {
|
||||
client := fake.NewSimpleClientset()
|
||||
config := &kubevip.Config{
|
||||
NodeName: "node-a",
|
||||
Annotations: "kube-vip.io",
|
||||
}
|
||||
|
||||
if err := annotationsWatcher(context.Background(), client, client, config); err == nil {
|
||||
t.Fatal("annotationsWatcher() error = nil, want empty node-list error")
|
||||
}
|
||||
}
|
||||
@@ -4,6 +4,8 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
log "log/slog"
|
||||
"net"
|
||||
"strconv"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
|
||||
@@ -53,9 +55,14 @@ func (b *BGP) Configure(ctx context.Context, _ *sync.WaitGroup) error {
|
||||
|
||||
log.Info("Starting the BGP server to advertise VIP routes to BGP peers")
|
||||
if err := b.bgpServer.Start(ctx, func(p *apiutil.WatchEventMessage_PeerEvent) {
|
||||
if p.Type != apiutil.PEER_EVENT_STATE {
|
||||
return
|
||||
}
|
||||
|
||||
ipaddr := p.Peer.State.NeighborAddress.String()
|
||||
port := uint64(179)
|
||||
peerDescription := fmt.Sprintf("%s:%d", ipaddr, port)
|
||||
|
||||
port := 179
|
||||
peerDescription := net.JoinHostPort(ipaddr, strconv.Itoa(port))
|
||||
|
||||
for stateName, stateValue := range api.PeerState_SessionState_value {
|
||||
metricValue := 0.0
|
||||
@@ -106,6 +113,10 @@ func (b *BGP) StartServices(ctx context.Context) error {
|
||||
if err := b.PerServiceLeader(ctx, false); err != nil {
|
||||
return err
|
||||
}
|
||||
} else if b.config.EnableLeaderElection {
|
||||
log.Warn("leader election is enabled, only the elected leader will advertise service VIPs; unset enable_leader_election to keep advertising from every node (ECMP)",
|
||||
"lease", b.config.ServicesLeaseName)
|
||||
b.GlobalLeader(ctx, b.config.ServicesLeaseName)
|
||||
} else {
|
||||
if err := b.ServicesNoLeader(ctx); err != nil {
|
||||
return err
|
||||
@@ -114,10 +125,6 @@ func (b *BGP) StartServices(ctx context.Context) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (b *BGP) ServicesGlobalLeader(ctx context.Context, id string) {
|
||||
// NOT IMPLEMENTED
|
||||
}
|
||||
|
||||
func (b *BGP) Name() string {
|
||||
return "BGP"
|
||||
}
|
||||
|
||||
@@ -17,6 +17,7 @@ import (
|
||||
"github.com/kube-vip/kube-vip/pkg/node"
|
||||
"github.com/kube-vip/kube-vip/pkg/route"
|
||||
"github.com/kube-vip/kube-vip/pkg/services"
|
||||
"github.com/kube-vip/kube-vip/pkg/vip"
|
||||
"k8s.io/client-go/kubernetes"
|
||||
)
|
||||
|
||||
@@ -96,7 +97,15 @@ func (c *Common) GlobalLeader(ctx context.Context, leaseName string) {
|
||||
})
|
||||
}
|
||||
|
||||
c.runGlobalElection(servicesCtx, c, leaseName, c.config, c.electionMgr)
|
||||
var vips []string
|
||||
if c.svcProcessor != nil {
|
||||
var err error
|
||||
vips, err = c.svcProcessor.ElectionVIPs(servicesCtx)
|
||||
if err != nil {
|
||||
log.Warn("unable to list Service VIPs for Lease metadata", "err", err)
|
||||
}
|
||||
}
|
||||
c.runGlobalElection(servicesCtx, c, leaseName, c.config, c.electionMgr, vips)
|
||||
}
|
||||
|
||||
func (c *Common) ServicesNoLeader(ctx context.Context) error {
|
||||
@@ -155,7 +164,7 @@ func (c *Common) OnNewLeader(identity string) {
|
||||
}
|
||||
|
||||
func (c *Common) runGlobalElection(ctx context.Context, a election.Actions, leaseName string,
|
||||
config *kubevip.Config, electionManager *election.Manager) {
|
||||
config *kubevip.Config, electionManager *election.Manager, vips []string) {
|
||||
|
||||
log.Debug("starting global election")
|
||||
ns, leaseName := lease.NamespaceName(leaseName, config)
|
||||
@@ -163,99 +172,73 @@ func (c *Common) runGlobalElection(ctx context.Context, a election.Actions, leas
|
||||
leaseID := lease.NewID(config.LeaderElectionType, ns, leaseName)
|
||||
objectName := lease.ObjectName(leaseID, "svcs0")
|
||||
|
||||
// objLease, isNew, isSharedLease := c.leaseMgr.Add(leaseID, objectName)
|
||||
objLease, _ := c.leaseMgr.Acquire(context.Background(), leaseID, objectName)
|
||||
defer c.leaseMgr.Delete(leaseID, objectName, objLease)
|
||||
electionCtx, cancelElection := objLease.NewElectionContext(ctx)
|
||||
defer cancelElection()
|
||||
|
||||
objLease := c.leaseMgr.Add(ctx, leaseID)
|
||||
isNew := objLease.Add(objectName)
|
||||
|
||||
// this service was already processed so we do not need to do anything
|
||||
if !isNew {
|
||||
log.Debug("this election was already done, waiting for it to finish", "lease", c.config.ServicesLeaseName)
|
||||
// Wait for either the service context or lease context to be done
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
// Service was deleted
|
||||
c.leaseMgr.Delete(leaseID, objectName)
|
||||
case <-objLease.Ctx.Done():
|
||||
// Leader election ended (leadership lost or context cancelled)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
objLease.Lock()
|
||||
|
||||
defer func() {
|
||||
objLease.Unlock()
|
||||
}()
|
||||
|
||||
if objLease.Elected.Load() {
|
||||
objLease.Unlock()
|
||||
for !objLease.BeginElection() {
|
||||
log.Debug("this election was already done, shared lease", "lease", leaseID.Name())
|
||||
leaderGeneration, elected := objLease.WaitForLeaderGeneration(electionCtx)
|
||||
if !elected {
|
||||
if electionCtx.Err() != nil {
|
||||
return
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// wait for leader election to start or context to be done
|
||||
select {
|
||||
case <-objLease.Started:
|
||||
case <-objLease.Ctx.Done():
|
||||
// Lease was cancelled (e.g., leader election ended), return immediately
|
||||
// This allows the restart loop to create a fresh lease
|
||||
log.Debug("lease context cancelled before leader election started", "lease", leaseID.Name())
|
||||
leaderCtx, cancelLeader := context.WithCancel(electionCtx)
|
||||
wg := sync.WaitGroup{}
|
||||
wg.Go(func() {
|
||||
a.OnStartedLeading(leaderCtx)
|
||||
})
|
||||
objLease.WaitForElectionEndAfter(electionCtx, leaderGeneration)
|
||||
cancelLeader()
|
||||
wg.Wait()
|
||||
if electionCtx.Err() != nil {
|
||||
return
|
||||
}
|
||||
|
||||
a.OnStartedLeading(objLease.Ctx)
|
||||
|
||||
log.Debug("waiting for lease to finish", "lease", leaseID.Name())
|
||||
// wait for leaderelection to be finished
|
||||
<-objLease.Ctx.Done()
|
||||
|
||||
// we can do cleanup here
|
||||
a.OnStoppedLeading()
|
||||
|
||||
log.Error("lost leadership, restarting kube-vip", "lease", leaseID.Name())
|
||||
c.killFunc()
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
// For new leases (not shared), ensure cleanup when the leader election ends
|
||||
// This is critical for the restartable service watcher to be able to restart
|
||||
// the leader election after leadership loss
|
||||
defer func() {
|
||||
// Delete the lease from the manager so subsequent calls can create a fresh lease
|
||||
// This handles the case where leader election ends due to:
|
||||
// 1. Leadership loss (e.g., network timeout)
|
||||
// 2. Context cancellation
|
||||
// 3. Any other reason RunOrDie returns
|
||||
c.leaseMgr.Delete(leaseID, objectName)
|
||||
}()
|
||||
|
||||
wg := sync.WaitGroup{}
|
||||
defer objLease.ElectionStopped()
|
||||
defer wg.Wait()
|
||||
|
||||
run := &election.RunConfig{
|
||||
Config: config,
|
||||
LeaseID: leaseID,
|
||||
LeaseAnnotations: map[string]string{},
|
||||
VIPs: vips,
|
||||
Mgr: electionManager,
|
||||
OnStartedLeading: func(ctx context.Context) {
|
||||
objLease.ElectionStarted()
|
||||
wg.Go(func() {
|
||||
objLease.Elected.Store(true)
|
||||
objLease.Unlock()
|
||||
close(objLease.Started)
|
||||
a.OnStartedLeading(ctx)
|
||||
metrics.LeaderTransitionsTotal.WithLabelValues(leaseID.Name()).Inc()
|
||||
metrics.IsLeader.WithLabelValues(config.NodeName, leaseID.Name()).Set(1)
|
||||
})
|
||||
},
|
||||
OnStoppedLeading: func() {
|
||||
objLease.Elected.Store(false)
|
||||
objLease.ElectionStopped()
|
||||
a.OnStoppedLeading()
|
||||
metrics.IsLeader.WithLabelValues(config.NodeName, leaseID.Name()).Set(0)
|
||||
},
|
||||
OnNewLeader: a.OnNewLeader,
|
||||
}
|
||||
|
||||
if err := election.RunOrDie(ctx, run, config); err != nil {
|
||||
if err := election.RunOrDie(electionCtx, run, config); err != nil {
|
||||
log.Error("leaderelection failed", "err", err, "id", config.NodeName, "name", leaseID.Name())
|
||||
}
|
||||
}
|
||||
|
||||
func controlPlaneElectionVIPs(config *kubevip.Config) []string {
|
||||
configured := config.VIP
|
||||
if config.Address != "" {
|
||||
configured = config.Address
|
||||
}
|
||||
return vip.Split(configured)
|
||||
}
|
||||
|
||||
115
pkg/manager/worker/common_test.go
Normal file
115
pkg/manager/worker/common_test.go
Normal file
@@ -0,0 +1,115 @@
|
||||
package worker
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
)
|
||||
|
||||
type sharedElectionActions struct {
|
||||
started chan struct{}
|
||||
stopped chan struct{}
|
||||
}
|
||||
|
||||
func (a *sharedElectionActions) OnStartedLeading(ctx context.Context) {
|
||||
close(a.started)
|
||||
<-ctx.Done()
|
||||
}
|
||||
|
||||
func (a *sharedElectionActions) OnStoppedLeading() {
|
||||
close(a.stopped)
|
||||
}
|
||||
|
||||
func (a *sharedElectionActions) OnNewLeader(string) {}
|
||||
|
||||
func TestGlobalElectionFollowsSharedLeaseLeadership(t *testing.T) {
|
||||
config := &kubevip.Config{KubernetesLeaderElection: kubevip.KubernetesLeaderElection{LeaseName: "default/shared"}}
|
||||
leaseID := lease.NewID(config.LeaderElectionType, "default", "shared")
|
||||
leaseMgr := lease.NewManager()
|
||||
sharedLease, _ := leaseMgr.Acquire(context.Background(), leaseID, "service")
|
||||
if !sharedLease.BeginElection() {
|
||||
t.Fatal("Service election did not start")
|
||||
}
|
||||
sharedLease.ElectionStarted()
|
||||
|
||||
actions := &sharedElectionActions{started: make(chan struct{}), stopped: make(chan struct{})}
|
||||
var killed atomic.Bool
|
||||
common := &Common{config: config, leaseMgr: leaseMgr, killFunc: func() { killed.Store(true) }}
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
common.runGlobalElection(context.Background(), actions, config.LeaseName, config, nil, nil)
|
||||
close(done)
|
||||
}()
|
||||
select {
|
||||
case <-actions.started:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("global election follower did not activate")
|
||||
}
|
||||
|
||||
sharedLease.ElectionStopped()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("global election follower did not stop after leadership ended")
|
||||
}
|
||||
select {
|
||||
case <-actions.stopped:
|
||||
default:
|
||||
t.Fatal("global election follower did not run leadership cleanup")
|
||||
}
|
||||
if !killed.Load() {
|
||||
t.Fatal("global election follower did not request restart after leadership loss")
|
||||
}
|
||||
if sharedLease.Ctx.Err() != nil || leaseMgr.Get(leaseID) != sharedLease {
|
||||
t.Fatal("global election follower cancelled the surviving Service lease")
|
||||
}
|
||||
leaseMgr.Delete(leaseID, "service", sharedLease)
|
||||
}
|
||||
|
||||
func TestGlobalElectionFollowerShutdownIsNotLeadershipLoss(t *testing.T) {
|
||||
config := &kubevip.Config{KubernetesLeaderElection: kubevip.KubernetesLeaderElection{LeaseName: "default/shared"}}
|
||||
leaseID := lease.NewID(config.LeaderElectionType, "default", "shared")
|
||||
leaseMgr := lease.NewManager()
|
||||
sharedLease, _ := leaseMgr.Acquire(context.Background(), leaseID, "service")
|
||||
if !sharedLease.BeginElection() {
|
||||
t.Fatal("Service election did not start")
|
||||
}
|
||||
sharedLease.ElectionStarted()
|
||||
|
||||
actions := &sharedElectionActions{started: make(chan struct{}), stopped: make(chan struct{})}
|
||||
var killed atomic.Bool
|
||||
common := &Common{config: config, leaseMgr: leaseMgr, killFunc: func() { killed.Store(true) }}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
common.runGlobalElection(ctx, actions, config.LeaseName, config, nil, nil)
|
||||
close(done)
|
||||
}()
|
||||
select {
|
||||
case <-actions.started:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("global election follower did not activate")
|
||||
}
|
||||
cancel()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("global election follower did not stop with its parent context")
|
||||
}
|
||||
select {
|
||||
case <-actions.stopped:
|
||||
t.Fatal("graceful follower shutdown was reported as leadership loss")
|
||||
default:
|
||||
}
|
||||
if killed.Load() {
|
||||
t.Fatal("graceful follower shutdown requested a process restart")
|
||||
}
|
||||
if sharedLease.Ctx.Err() != nil || leaseMgr.Get(leaseID) != sharedLease {
|
||||
t.Fatal("global follower shutdown cancelled the surviving Service lease")
|
||||
}
|
||||
leaseMgr.Delete(leaseID, "service", sharedLease)
|
||||
}
|
||||
@@ -43,7 +43,11 @@ func (t *Table) Configure(ctx context.Context, wg *sync.WaitGroup) error {
|
||||
if t.config.CleanRoutingTable {
|
||||
wg.Go(func() {
|
||||
// we assume that after 10s all services should be configured so we can delete redundant routes
|
||||
time.Sleep(time.Second * 10)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-time.After(10 * time.Second):
|
||||
}
|
||||
if err := t.cleanRoutes(); err != nil {
|
||||
log.Error("error checking for old routes", "err", err)
|
||||
}
|
||||
@@ -100,7 +104,7 @@ func (t *Table) cleanRoutes() error {
|
||||
if t.config.EnableControlPlane {
|
||||
found = (routes[i].Dst.IP.String() == t.config.Address)
|
||||
} else {
|
||||
found = t.routeMgr.Check(routes[i].String())
|
||||
found = t.routeMgr.Check(vip.NetlinkHash(&(routes[i])))
|
||||
}
|
||||
|
||||
if !found {
|
||||
|
||||
47
pkg/manager/worker/table_ctx_test.go
Normal file
47
pkg/manager/worker/table_ctx_test.go
Normal file
@@ -0,0 +1,47 @@
|
||||
package worker
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
)
|
||||
|
||||
func TestConfigureCleanRoutingTableStopsWithContext(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
var wg sync.WaitGroup
|
||||
table := &Table{Common: Common{
|
||||
config: &kubevip.Config{
|
||||
CleanRoutingTable: true,
|
||||
EnableControlPlane: true,
|
||||
Address: "10.254.254.254",
|
||||
RoutingTableID: 0x7fffffff,
|
||||
RoutingProtocol: 255,
|
||||
},
|
||||
mutex: &sync.Mutex{},
|
||||
}}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
|
||||
if err := table.Configure(ctx, &wg); err != nil {
|
||||
t.Fatalf("Configure returned an error: %v", err)
|
||||
}
|
||||
cancel()
|
||||
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
wg.Wait()
|
||||
close(done)
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(500 * time.Millisecond):
|
||||
// DEFECT: pkg/manager/worker/table.go:43-48 uses an unconditional
|
||||
// 10-second sleep for cleanRoutingTable and ignores the canceled RT
|
||||
// worker context, delaying shutdown/recovery.
|
||||
t.Fatal("cleanRoutingTable worker did not stop after context cancellation")
|
||||
}
|
||||
}
|
||||
@@ -27,6 +27,8 @@ import (
|
||||
"k8s.io/client-go/kubernetes"
|
||||
)
|
||||
|
||||
const controlPlaneTunnelOwner = "control-plane"
|
||||
|
||||
type WireGuard struct {
|
||||
Common
|
||||
tunnelMgr *wireguard.TunnelManager
|
||||
@@ -94,7 +96,7 @@ func (w *WireGuard) StartControlPlane(ctx context.Context, electionManager *elec
|
||||
log.Error("no WireGuard tunnel configuration found for control plane VIP", "vip", w.config.VIP)
|
||||
return
|
||||
}
|
||||
w.runGlobalElection(ctx, w, w.config.LeaseName, w.config, electionManager)
|
||||
w.runGlobalElection(ctx, w, w.config.LeaseName, w.config, electionManager, controlPlaneElectionVIPs(w.config))
|
||||
}
|
||||
|
||||
func (w *WireGuard) ConfigureServices() {
|
||||
@@ -102,12 +104,15 @@ func (w *WireGuard) ConfigureServices() {
|
||||
}
|
||||
|
||||
func (w *WireGuard) StartServices(ctx context.Context) error {
|
||||
// WireGuard has no multipath mechanism, so every service must be advertised by
|
||||
// exactly one node: leader election (per-service or global) is required.
|
||||
if w.config.EnableServicesElection {
|
||||
log.Info("beginning watching services, leaderelection will happen for every service")
|
||||
err := w.svcProcessor.StartServicesWatchForLeaderElection(ctx, false)
|
||||
if err != nil {
|
||||
if err := w.svcProcessor.StartServicesWatchForLeaderElection(ctx, false); err != nil {
|
||||
return err
|
||||
}
|
||||
} else {
|
||||
w.GlobalLeader(ctx, w.config.ServicesLeaseName)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -118,10 +123,9 @@ func (w *WireGuard) Name() string {
|
||||
|
||||
func (w *WireGuard) OnStartedLeading(ctx context.Context) {
|
||||
// Bring up the WireGuard tunnel for control plane VIP
|
||||
err := w.tunnelMgr.BringUpTunnelForVIP(w.config.VIP)
|
||||
err := w.tunnelMgr.AcquireTunnelForVIP(w.config.VIP, controlPlaneTunnelOwner)
|
||||
if err != nil {
|
||||
log.Error("could not start wireguard tunnel for control plane", "vip", w.config.VIP, "err", err)
|
||||
_ = w.tunnelMgr.TearDownTunnelForVIP(w.config.VIP)
|
||||
w.killFunc()
|
||||
return
|
||||
}
|
||||
@@ -130,6 +134,7 @@ func (w *WireGuard) OnStartedLeading(ctx context.Context) {
|
||||
wg := w.tunnelMgr.GetTunnelForVIP(w.config.VIP)
|
||||
if wg == nil {
|
||||
log.Error("failed to get wireguard tunnel after bringing up", "vip", w.config.VIP)
|
||||
_ = w.tunnelMgr.ReleaseTunnelForVIP(w.config.VIP, controlPlaneTunnelOwner)
|
||||
w.killFunc()
|
||||
return
|
||||
}
|
||||
@@ -137,7 +142,7 @@ func (w *WireGuard) OnStartedLeading(ctx context.Context) {
|
||||
tunnelConfig := w.tunnelMgr.GetConfigForVIP(w.config.VIP)
|
||||
if tunnelConfig == nil {
|
||||
log.Error("failed to get tunnel configuration", "vip", w.config.VIP)
|
||||
_ = w.tunnelMgr.TearDownTunnelForVIP(w.config.VIP)
|
||||
_ = w.tunnelMgr.ReleaseTunnelForVIP(w.config.VIP, controlPlaneTunnelOwner)
|
||||
w.killFunc()
|
||||
return
|
||||
}
|
||||
@@ -147,12 +152,6 @@ func (w *WireGuard) OnStartedLeading(ctx context.Context) {
|
||||
w.endpointWatcherWg.Go(func() {
|
||||
w.watchKubernetesEndpoints(w.endpointWatcherCtx, tunnelConfig)
|
||||
})
|
||||
|
||||
if w.config.EnableServices && !w.config.EnableServicesElection {
|
||||
if err := w.svcProcessor.ServicesWatcher(ctx, services.NewCallback(w.svcProcessor.SyncServices, false), false); err != nil {
|
||||
log.Error("failed to start services watcher", "err", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// watchKubernetesEndpoints watches the kubernetes service EndpointSlices for changes
|
||||
@@ -185,8 +184,15 @@ func (w *WireGuard) watchKubernetesEndpoints(ctx context.Context, tunnelConfig *
|
||||
|
||||
switch event.Type {
|
||||
case watch.Added, watch.Modified, watch.Deleted:
|
||||
if err := provider.LoadObject(event.Object, func() {}); err != nil {
|
||||
log.Error("failed to load endpoint object", "err", err)
|
||||
// A deleted slice has to be dropped so it stops counting toward the endpoint set.
|
||||
var err error
|
||||
if event.Type == watch.Deleted {
|
||||
err = provider.DeleteObject(event.Object)
|
||||
} else {
|
||||
err = provider.LoadObject(event.Object, func() {})
|
||||
}
|
||||
if err != nil {
|
||||
log.Error("failed to update endpoint object", "eventType", event.Type, "err", err)
|
||||
continue
|
||||
}
|
||||
endpoints, _ := provider.GetAllEndpoints()
|
||||
@@ -273,6 +279,6 @@ func (w *WireGuard) OnNewLeader(identity string) {
|
||||
return
|
||||
}
|
||||
// safety check - tear down tunnel if we're not the leader
|
||||
_ = w.tunnelMgr.TearDownTunnelForVIP(w.config.VIP)
|
||||
_ = w.tunnelMgr.ReleaseTunnelForVIP(w.config.VIP, controlPlaneTunnelOwner)
|
||||
log.Info("new leader elected", "id", identity)
|
||||
}
|
||||
|
||||
@@ -35,6 +35,21 @@ var (
|
||||
prometheus.GaugeOpts{Name: "kube_vip_is_leader", Help: "1 if this node currently holds the lease"},
|
||||
[]string{"node", "lease_name"},
|
||||
)
|
||||
ServiceElectionLoops = prometheus.NewGaugeVec(
|
||||
prometheus.GaugeOpts{Name: "kube_vip_service_election_loops",
|
||||
Help: "Live per-service leader election restart loops on this node; more than 1 per service means loops leaked"},
|
||||
[]string{"namespace", "name"},
|
||||
)
|
||||
ServiceElectionAttemptsTotal = prometheus.NewCounterVec(
|
||||
prometheus.CounterOpts{Name: "kube_vip_service_election_attempts_total",
|
||||
Help: "Election attempts made by the per-service leader election restart loop"},
|
||||
[]string{"namespace", "name"},
|
||||
)
|
||||
ServiceElectionErrorsTotal = prometheus.NewCounterVec(
|
||||
prometheus.CounterOpts{Name: "kube_vip_service_election_errors_total",
|
||||
Help: "Per-service leader election failures by reason"},
|
||||
[]string{"namespace", "name", "reason"},
|
||||
)
|
||||
|
||||
// This is a prometheus gauge indicating the state of the sessions.
|
||||
// 1 means "ESTABLISHED", 0 means "NOT ESTABLISHED"
|
||||
@@ -61,6 +76,9 @@ func RegisterPrometheusMetrics() {
|
||||
ServiceReconcileDuration,
|
||||
LeaderTransitionsTotal,
|
||||
IsLeader,
|
||||
ServiceElectionLoops,
|
||||
ServiceElectionAttemptsTotal,
|
||||
ServiceElectionErrorsTotal,
|
||||
BGPSessionInfoGauge,
|
||||
BuildInfo,
|
||||
CountServiceWatchEvent,
|
||||
|
||||
@@ -1,19 +1,19 @@
|
||||
package networkinterface
|
||||
|
||||
import (
|
||||
log "log/slog"
|
||||
"sync"
|
||||
|
||||
"github.com/vishvananda/netlink"
|
||||
)
|
||||
|
||||
type Manager struct {
|
||||
lock sync.Mutex
|
||||
interfaces map[string]*Link
|
||||
}
|
||||
|
||||
type Link struct {
|
||||
Lock sync.Mutex
|
||||
Intf netlink.Link
|
||||
mu sync.Mutex
|
||||
intf netlink.Link
|
||||
}
|
||||
|
||||
func NewManager() *Manager {
|
||||
@@ -23,19 +23,37 @@ func NewManager() *Manager {
|
||||
}
|
||||
|
||||
func (m *Manager) Get(intf netlink.Link) *Link {
|
||||
if l, ok := m.interfaces[intf.Attrs().Name]; ok {
|
||||
updated, err := netlink.LinkByName(l.Intf.Attrs().Name)
|
||||
if err != nil {
|
||||
log.Error("failed to get interface %q: %w", l.Intf.Attrs().Name, err)
|
||||
return nil
|
||||
}
|
||||
l.Intf = updated
|
||||
return l
|
||||
if intf == nil || intf.Attrs() == nil {
|
||||
return nil
|
||||
}
|
||||
result := &Link{
|
||||
Intf: intf,
|
||||
attrs := intf.Attrs()
|
||||
|
||||
m.lock.Lock()
|
||||
defer m.lock.Unlock()
|
||||
if link, ok := m.interfaces[attrs.Name]; ok {
|
||||
link.replace(intf)
|
||||
return link
|
||||
}
|
||||
|
||||
m.interfaces[intf.Attrs().Name] = result
|
||||
return result
|
||||
link := &Link{intf: intf}
|
||||
m.interfaces[attrs.Name] = link
|
||||
return link
|
||||
}
|
||||
|
||||
func (l *Link) WithInterface(run func(netlink.Link) error) error {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
return run(l.intf)
|
||||
}
|
||||
|
||||
func (l *Link) replace(intf netlink.Link) {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
l.intf = intf
|
||||
}
|
||||
|
||||
func (m *Manager) Len() int {
|
||||
m.lock.Lock()
|
||||
defer m.lock.Unlock()
|
||||
return len(m.interfaces)
|
||||
}
|
||||
|
||||
65
pkg/networkinterface/networkinterface_instance_test.go
Normal file
65
pkg/networkinterface/networkinterface_instance_test.go
Normal file
@@ -0,0 +1,65 @@
|
||||
package networkinterface_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/arp"
|
||||
"github.com/kube-vip/kube-vip/pkg/instance"
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/networkinterface"
|
||||
"github.com/kube-vip/kube-vip/pkg/node/noop"
|
||||
"github.com/kube-vip/kube-vip/pkg/route"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
)
|
||||
|
||||
func TestManagerReconstructsProductionInstanceConcurrently(t *testing.T) {
|
||||
config := &kubevip.Config{Interface: "lo", ServicesInterface: "lo", VIPSubnet: "32", DisableServiceUpdates: true}
|
||||
manager := networkinterface.NewManager()
|
||||
service := &v1.Service{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "service", Namespace: "default", UID: "service"},
|
||||
Spec: v1.ServiceSpec{LoadBalancerIP: "192.0.2.10"},
|
||||
}
|
||||
|
||||
start := make(chan struct{})
|
||||
var ready sync.WaitGroup
|
||||
ready.Add(2)
|
||||
results := make(chan struct {
|
||||
instance *instance.Instance
|
||||
err error
|
||||
}, 2)
|
||||
for range 2 {
|
||||
go func() {
|
||||
ready.Done()
|
||||
<-start
|
||||
instanceConfig := *config
|
||||
created, err := instance.NewInstance(context.Background(), service.DeepCopy(), &instanceConfig, manager,
|
||||
arp.NewManager(&instanceConfig), route.NewManager(), noop.NewManager(), &sync.WaitGroup{})
|
||||
results <- struct {
|
||||
instance *instance.Instance
|
||||
err error
|
||||
}{created, err}
|
||||
}()
|
||||
}
|
||||
ready.Wait()
|
||||
close(start)
|
||||
for range 2 {
|
||||
select {
|
||||
case result := <-results:
|
||||
if result.err != nil {
|
||||
t.Fatalf("NewInstance() error = %v", result.err)
|
||||
}
|
||||
if len(result.instance.Clusters) != 1 {
|
||||
t.Fatalf("cluster count = %d, want 1", len(result.instance.Clusters))
|
||||
}
|
||||
case <-time.After(3 * time.Second):
|
||||
t.Fatal("timed out waiting for concurrent NewInstance calls")
|
||||
}
|
||||
}
|
||||
if got := manager.Len(); got != 1 {
|
||||
t.Fatalf("cached link count = %d, want 1", got)
|
||||
}
|
||||
}
|
||||
74
pkg/networkinterface/networkinterface_test.go
Normal file
74
pkg/networkinterface/networkinterface_test.go
Normal file
@@ -0,0 +1,74 @@
|
||||
package networkinterface
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"github.com/vishvananda/netlink"
|
||||
)
|
||||
|
||||
func TestManagerGetReplacesChangedInterfaceIndex(t *testing.T) {
|
||||
manager := NewManager()
|
||||
firstInterface := dummyLink("eth0", 1)
|
||||
first := manager.Get(firstInterface)
|
||||
|
||||
if got := manager.Get(dummyLink("eth0", 1)); got != first {
|
||||
t.Fatal("Get returned a new link for the same interface generation")
|
||||
}
|
||||
|
||||
secondInterface := dummyLink("eth0", 2)
|
||||
second := manager.Get(secondInterface)
|
||||
if second != first {
|
||||
t.Fatal("Get replaced the shared link after the interface index changed")
|
||||
}
|
||||
var current netlink.Link
|
||||
if err := first.WithInterface(func(intf netlink.Link) error {
|
||||
current = intf
|
||||
return nil
|
||||
}); err != nil {
|
||||
t.Fatalf("WithInterface() error = %v", err)
|
||||
}
|
||||
if current != secondInterface {
|
||||
t.Fatal("Get did not retain the new link generation")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerGetConcurrent(t *testing.T) {
|
||||
manager := NewManager()
|
||||
interfaces := []netlink.Link{dummyLink("eth0", 1), dummyLink("eth1", 2)}
|
||||
var wg sync.WaitGroup
|
||||
results := make(chan struct {
|
||||
index int
|
||||
link *Link
|
||||
}, 64)
|
||||
for index := range cap(results) {
|
||||
interfaceIndex := index % len(interfaces)
|
||||
wg.Go(func() {
|
||||
results <- struct {
|
||||
index int
|
||||
link *Link
|
||||
}{index: interfaceIndex, link: manager.Get(interfaces[interfaceIndex])}
|
||||
})
|
||||
}
|
||||
wg.Wait()
|
||||
close(results)
|
||||
|
||||
var cached [2]*Link
|
||||
for result := range results {
|
||||
if result.link == nil {
|
||||
t.Fatal("concurrent interface lookup returned nil")
|
||||
}
|
||||
if cached[result.index] == nil {
|
||||
cached[result.index] = result.link
|
||||
} else if result.link != cached[result.index] {
|
||||
t.Fatalf("interface %d produced multiple cached Link objects", result.index)
|
||||
}
|
||||
}
|
||||
if cached[0] == cached[1] {
|
||||
t.Fatal("different interfaces shared one cached Link object")
|
||||
}
|
||||
}
|
||||
|
||||
func dummyLink(name string, index int) netlink.Link {
|
||||
return &netlink.Dummy{LinkAttrs: netlink.LinkAttrs{Name: name, Index: index}}
|
||||
}
|
||||
@@ -52,7 +52,7 @@ const (
|
||||
// For the fwmark, we add an offset (0x10000) to avoid collision with WireGuard's own fwmark.
|
||||
// WireGuard sets fwmark=listenPort on its own UDP packets to the peer, so we must use
|
||||
// a different mark value for our connmark-based policy routing.
|
||||
const connmarkOffset = 0x10000 // 65536 - added to listenPort to get our connmark value
|
||||
const fwmarkOffset = 0x10000 // 65536 - added to listenPort to get our fwmark value
|
||||
|
||||
func ApplySNAT(podIP, vipIP, service, destinationPorts string, ignoreCIDR []string, allowCIDR []string, IPv6 bool) error {
|
||||
return ApplySNATWithTable(podIP, vipIP, service, destinationPorts, ignoreCIDR, allowCIDR, IPv6, "")
|
||||
@@ -973,7 +973,11 @@ func EnsureTunnelInfrastructure(wgIf string, vipIP string, IPv6 bool, tunnelList
|
||||
conn.AddRule(masqRule)
|
||||
|
||||
// Add connmark rules
|
||||
fwmark := uint32(tunnelListenPort) + connmarkOffset //nolint:gosec
|
||||
// We use the fwmarkOffset NOT for the connmark, as the results in conn marks with 0x1xxxx
|
||||
// As this 1 is now in the third least-significant byte range, we risk interfering with more software
|
||||
// than necessary. One example is tailscale https://github.com/tailscale/tailscale/pull/19725
|
||||
connmark := uint32(tunnelListenPort) //nolint:gosec
|
||||
fwmark := connmark + fwmarkOffset
|
||||
|
||||
connmarkInRule := &nftables.Rule{
|
||||
Table: table,
|
||||
@@ -981,7 +985,7 @@ func EnsureTunnelInfrastructure(wgIf string, vipIP string, IPv6 bool, tunnelList
|
||||
Exprs: []expr.Any{
|
||||
&expr.Meta{Key: expr.MetaKeyIIFNAME, Register: 1},
|
||||
&expr.Cmp{Op: expr.CmpOpEq, Register: 1, Data: ifname(wgIf)},
|
||||
&expr.Immediate{Register: 1, Data: binaryutil.NativeEndian.PutUint32(fwmark)},
|
||||
&expr.Immediate{Register: 1, Data: binaryutil.NativeEndian.PutUint32(connmark)},
|
||||
&expr.Ct{Key: expr.CtKeyMARK, Register: 1, SourceRegister: true},
|
||||
},
|
||||
}
|
||||
@@ -994,7 +998,8 @@ func EnsureTunnelInfrastructure(wgIf string, vipIP string, IPv6 bool, tunnelList
|
||||
&expr.Meta{Key: expr.MetaKeyIIFNAME, Register: 1},
|
||||
&expr.Cmp{Op: expr.CmpOpNeq, Register: 1, Data: ifname(wgIf)},
|
||||
&expr.Ct{Key: expr.CtKeyMARK, Register: 1},
|
||||
&expr.Cmp{Op: expr.CmpOpEq, Register: 1, Data: binaryutil.NativeEndian.PutUint32(fwmark)},
|
||||
&expr.Cmp{Op: expr.CmpOpEq, Register: 1, Data: binaryutil.NativeEndian.PutUint32(connmark)},
|
||||
&expr.Immediate{Register: 1, Data: binaryutil.NativeEndian.PutUint32(fwmark)},
|
||||
&expr.Meta{Key: expr.MetaKeyMARK, SourceRegister: true, Register: 1},
|
||||
},
|
||||
}
|
||||
@@ -1005,7 +1010,8 @@ func EnsureTunnelInfrastructure(wgIf string, vipIP string, IPv6 bool, tunnelList
|
||||
Chain: mangleOutChain,
|
||||
Exprs: []expr.Any{
|
||||
&expr.Ct{Key: expr.CtKeyMARK, Register: 1},
|
||||
&expr.Cmp{Op: expr.CmpOpEq, Register: 1, Data: binaryutil.NativeEndian.PutUint32(fwmark)},
|
||||
&expr.Cmp{Op: expr.CmpOpEq, Register: 1, Data: binaryutil.NativeEndian.PutUint32(connmark)},
|
||||
&expr.Immediate{Register: 1, Data: binaryutil.NativeEndian.PutUint32(fwmark)},
|
||||
&expr.Meta{Key: expr.MetaKeyMARK, SourceRegister: true, Register: 1},
|
||||
},
|
||||
}
|
||||
@@ -1392,7 +1398,7 @@ func SetupPolicyRouting(wgIf string, listenPort int) error {
|
||||
return fmt.Errorf("failed to get interface %s: %w", wgIf, err)
|
||||
}
|
||||
|
||||
mark := uint32(listenPort) + connmarkOffset //nolint:gosec // Port range validated
|
||||
mark := uint32(listenPort) + fwmarkOffset //nolint:gosec // Port range validated
|
||||
table := listenPort
|
||||
|
||||
rule := netlink.NewRule()
|
||||
@@ -1558,7 +1564,7 @@ func bypassRpfilterForInterface(wgIf string) error {
|
||||
|
||||
// CleanupPolicyRouting removes the policy routing rule for a specific tunnel
|
||||
func CleanupPolicyRouting(listenPort int) error {
|
||||
mark := uint32(listenPort) + connmarkOffset //nolint:gosec // Port range validated
|
||||
mark := uint32(listenPort) + fwmarkOffset //nolint:gosec // Port range validated
|
||||
table := listenPort
|
||||
|
||||
rule := netlink.NewRule()
|
||||
|
||||
@@ -49,9 +49,6 @@ func (m *Manager) Add(object string, r route, precheck, update bool) error {
|
||||
itm, exists := m.tracker[key]
|
||||
|
||||
if !exists {
|
||||
m.tracker[key] = newItem(r)
|
||||
itm = m.tracker[key]
|
||||
|
||||
added, err := r.AddRoute(precheck)
|
||||
if err != nil {
|
||||
if update && errors.Is(err, syscall.EEXIST) && update {
|
||||
@@ -70,9 +67,13 @@ func (m *Manager) Add(object string, r route, precheck, update bool) error {
|
||||
}
|
||||
}
|
||||
|
||||
itm = newItem(r)
|
||||
m.tracker[key] = itm
|
||||
|
||||
if added {
|
||||
log.Debug("[RT] added route", "path", key, "object", object)
|
||||
}
|
||||
m.tracker[key] = itm
|
||||
}
|
||||
|
||||
itm.objects[object] = true
|
||||
@@ -112,6 +113,8 @@ func (m *Manager) Delete(object string, r route) error {
|
||||
}
|
||||
|
||||
func (m *Manager) Clear() {
|
||||
m.mtx.Lock()
|
||||
defer m.mtx.Unlock()
|
||||
for _, itm := range m.tracker {
|
||||
if err := itm.route.DeleteRoute(); err != nil {
|
||||
log.Warn("[RT] failed to delete route", "err", err.Error())
|
||||
@@ -121,6 +124,8 @@ func (m *Manager) Clear() {
|
||||
}
|
||||
|
||||
func (m *Manager) Check(key string) bool {
|
||||
m.mtx.Lock()
|
||||
defer m.mtx.Unlock()
|
||||
_, exists := m.tracker[key]
|
||||
return exists
|
||||
}
|
||||
|
||||
29
pkg/route/manager_atomic_test.go
Normal file
29
pkg/route/manager_atomic_test.go
Normal file
@@ -0,0 +1,29 @@
|
||||
package route
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"testing"
|
||||
)
|
||||
|
||||
var errTransientRouteAdd = errors.New("transient route add failure")
|
||||
|
||||
func TestManagerRetriesRouteAfterInitialAddFailure(t *testing.T) {
|
||||
m := NewManager()
|
||||
r := &mockRoute{hash: "retry", addErr: errTransientRouteAdd}
|
||||
|
||||
if err := m.Add("service", r, false, true); !errors.Is(err, errTransientRouteAdd) {
|
||||
t.Fatalf("first add error = %v, want %v", err, errTransientRouteAdd)
|
||||
}
|
||||
|
||||
// A transient netlink failure (for example while the link is being recreated)
|
||||
// must not poison the in-memory tracker. The next reconciliation has to retry
|
||||
// the kernel operation.
|
||||
r.addErr = nil
|
||||
r.added = true
|
||||
if err := m.Add("service", r, false, true); err != nil {
|
||||
t.Fatalf("retry add failed: %v", err)
|
||||
}
|
||||
if r.addCalls != 2 {
|
||||
t.Fatalf("AddRoute called %d times, want 2", r.addCalls)
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,9 @@
|
||||
package route
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"sync"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -184,8 +186,50 @@ func Test_MultipleRoutesAddDel(t *testing.T) {
|
||||
})
|
||||
}
|
||||
|
||||
func TestAddFailureDoesNotTrackRoute(t *testing.T) {
|
||||
m := NewManager()
|
||||
r := &mockRoute{hash: "failed-route", addErr: errors.New("add failed")}
|
||||
|
||||
if err := m.Add("service", r, false, false); err == nil {
|
||||
t.Fatal("Add error = nil, want route failure")
|
||||
}
|
||||
if m.Check(r.RouteHash()) {
|
||||
t.Fatal("failed route was tracked")
|
||||
}
|
||||
|
||||
r.addErr = nil
|
||||
if err := m.Add("service", r, false, false); err != nil {
|
||||
t.Fatalf("retry Add error = %v", err)
|
||||
}
|
||||
if !m.Check(r.RouteHash()) {
|
||||
t.Fatal("successful retry was not tracked")
|
||||
}
|
||||
}
|
||||
|
||||
func TestClearAndCheckAreSafeWithRouteUpdates(t *testing.T) {
|
||||
manager := NewManager()
|
||||
route := &mockRoute{hash: "concurrent-route", added: true}
|
||||
if err := manager.Add("service", route, false, false); err != nil {
|
||||
t.Fatalf("Add() error = %v", err)
|
||||
}
|
||||
|
||||
var wg sync.WaitGroup
|
||||
for range 20 {
|
||||
wg.Go(func() {
|
||||
manager.Check(route.RouteHash())
|
||||
})
|
||||
}
|
||||
wg.Go(manager.Clear)
|
||||
wg.Wait()
|
||||
|
||||
if manager.Check(route.RouteHash()) {
|
||||
t.Fatal("route remained tracked after concurrent Clear")
|
||||
}
|
||||
}
|
||||
|
||||
type mockRoute struct {
|
||||
added bool
|
||||
addCalls int
|
||||
addErr error
|
||||
updated bool
|
||||
updateErr error
|
||||
@@ -195,6 +239,7 @@ type mockRoute struct {
|
||||
}
|
||||
|
||||
func (mr *mockRoute) AddRoute(_ bool) (bool, error) {
|
||||
mr.addCalls++
|
||||
return mr.added, mr.addErr
|
||||
}
|
||||
|
||||
|
||||
@@ -3,55 +3,178 @@ package servicecontext
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
)
|
||||
|
||||
type Context struct {
|
||||
Ctx context.Context
|
||||
Cancel context.CancelFunc
|
||||
IsWatched bool
|
||||
ConfiguredNetworks sync.Map
|
||||
EndpointsReady chan any
|
||||
epReady sync.Once
|
||||
Signalled atomic.Bool
|
||||
LeaderCancel context.CancelFunc
|
||||
Ctx context.Context
|
||||
Cancel context.CancelFunc
|
||||
ConfiguredNetworks sync.Map
|
||||
stateMutex sync.Mutex
|
||||
ready bool
|
||||
isWatched bool
|
||||
watchingStopped chan struct{}
|
||||
endpointsReady chan any
|
||||
endpointsLost chan any
|
||||
readinessGeneration uint64
|
||||
readinessOperations int
|
||||
readinessChanged *sync.Cond
|
||||
}
|
||||
|
||||
func New(ctx context.Context) *Context {
|
||||
// context and cancel stored for a future use, gosec linter disabled
|
||||
svcCtx, svcCancel := context.WithCancel(ctx) //nolint:gosec
|
||||
return &Context{
|
||||
Ctx: svcCtx,
|
||||
Cancel: svcCancel,
|
||||
EndpointsReady: make(chan any),
|
||||
serviceContext := &Context{
|
||||
Ctx: svcCtx,
|
||||
Cancel: svcCancel,
|
||||
endpointsReady: make(chan any),
|
||||
endpointsLost: make(chan any),
|
||||
readinessGeneration: 1,
|
||||
}
|
||||
serviceContext.readinessChanged = sync.NewCond(&serviceContext.stateMutex)
|
||||
return serviceContext
|
||||
}
|
||||
|
||||
func (ctx *Context) HasConfiguredNetworks() bool {
|
||||
cnt := 0
|
||||
ctx.ConfiguredNetworks.Range(func(_ any, _ any) bool {
|
||||
cnt++
|
||||
return cnt < 1
|
||||
})
|
||||
return cnt > 0
|
||||
// ReadinessState returns one readiness lifecycle. The ready channel is closed
|
||||
// when endpoints become usable and the lost channel is closed when that exact
|
||||
// generation is reset.
|
||||
func (ctx *Context) ReadinessState() (uint64, <-chan any, <-chan any, bool) {
|
||||
ctx.stateMutex.Lock()
|
||||
defer ctx.stateMutex.Unlock()
|
||||
return ctx.readinessGeneration, ctx.endpointsReady, ctx.endpointsLost, ctx.ready
|
||||
}
|
||||
|
||||
func (ctx *Context) IsNetworkConfigured(ip string) bool {
|
||||
_, exists := ctx.ConfiguredNetworks.Load(ip)
|
||||
return exists
|
||||
func (ctx *Context) IsReady() bool {
|
||||
ctx.stateMutex.Lock()
|
||||
defer ctx.stateMutex.Unlock()
|
||||
return ctx.ready
|
||||
}
|
||||
|
||||
// ReadinessGenerationCurrent reports whether generation is the current usable
|
||||
// endpoint generation for this Service context.
|
||||
func (ctx *Context) ReadinessGenerationCurrent(generation uint64) bool {
|
||||
ctx.stateMutex.Lock()
|
||||
defer ctx.stateMutex.Unlock()
|
||||
return ctx.ready && ctx.readinessGeneration == generation
|
||||
}
|
||||
|
||||
func (ctx *Context) ResetReadinessGeneration(generation uint64) bool {
|
||||
ctx.stateMutex.Lock()
|
||||
defer ctx.stateMutex.Unlock()
|
||||
if !ctx.ready || ctx.readinessGeneration != generation {
|
||||
return false
|
||||
}
|
||||
close(ctx.endpointsLost)
|
||||
ctx.readinessGeneration++
|
||||
ctx.endpointsReady = make(chan any)
|
||||
ctx.endpointsLost = make(chan any)
|
||||
ctx.ready = false
|
||||
for ctx.readinessOperations > 0 {
|
||||
ctx.readinessChanged.Wait()
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func (ctx *Context) SignalReadiness() {
|
||||
ctx.epReady.Do(func() {
|
||||
close(ctx.EndpointsReady)
|
||||
ctx.Signalled.Store(true)
|
||||
})
|
||||
ctx.stateMutex.Lock()
|
||||
defer ctx.stateMutex.Unlock()
|
||||
if ctx.ready {
|
||||
return
|
||||
}
|
||||
close(ctx.endpointsReady)
|
||||
ctx.ready = true
|
||||
}
|
||||
|
||||
func (ctx *Context) ResetReadiness() {
|
||||
if ctx.Signalled.Load() {
|
||||
ctx.EndpointsReady = make(chan any)
|
||||
ctx.epReady = sync.Once{}
|
||||
ctx.Signalled.Store(false)
|
||||
// WaitForReadiness reserves the first ready generation that remains current.
|
||||
// The caller must release the returned reservation after its datapath operation.
|
||||
func (ctx *Context) WaitForReadiness() (func(), bool) {
|
||||
for {
|
||||
generation, ready, _, isReady := ctx.ReadinessState()
|
||||
if !isReady {
|
||||
select {
|
||||
case <-ctx.Ctx.Done():
|
||||
return nil, false
|
||||
case <-ready:
|
||||
}
|
||||
}
|
||||
if release, acquired := ctx.AcquireReadinessGeneration(generation); acquired {
|
||||
return release, true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// AcquireReadinessGeneration reserves a ready generation while a caller starts
|
||||
// or stops datapath work. ResetReadinessGeneration waits for the returned
|
||||
// release function, preventing that work from outliving its endpoint state.
|
||||
func (ctx *Context) AcquireReadinessGeneration(generation uint64) (func(), bool) {
|
||||
if !ctx.acquireReadinessGeneration(generation) {
|
||||
return nil, false
|
||||
}
|
||||
|
||||
var releaseOnce sync.Once
|
||||
return func() {
|
||||
releaseOnce.Do(ctx.releaseReadinessGeneration)
|
||||
}, true
|
||||
}
|
||||
|
||||
func (ctx *Context) acquireReadinessGeneration(generation uint64) bool {
|
||||
ctx.stateMutex.Lock()
|
||||
defer ctx.stateMutex.Unlock()
|
||||
if ctx.Ctx.Err() != nil || !ctx.ready || ctx.readinessGeneration != generation {
|
||||
return false
|
||||
}
|
||||
ctx.readinessOperations++
|
||||
return true
|
||||
}
|
||||
|
||||
func (ctx *Context) releaseReadinessGeneration() {
|
||||
ctx.stateMutex.Lock()
|
||||
defer ctx.stateMutex.Unlock()
|
||||
ctx.readinessOperations--
|
||||
if ctx.readinessOperations == 0 {
|
||||
ctx.readinessChanged.Broadcast()
|
||||
}
|
||||
}
|
||||
|
||||
func (ctx *Context) StartWatching() bool {
|
||||
ctx.stateMutex.Lock()
|
||||
defer ctx.stateMutex.Unlock()
|
||||
if ctx.Ctx.Err() != nil || ctx.isWatched {
|
||||
return false
|
||||
}
|
||||
ctx.isWatched = true
|
||||
ctx.watchingStopped = make(chan struct{})
|
||||
return true
|
||||
}
|
||||
|
||||
func (ctx *Context) StopWatching() {
|
||||
ctx.stateMutex.Lock()
|
||||
defer ctx.stateMutex.Unlock()
|
||||
if !ctx.isWatched {
|
||||
return
|
||||
}
|
||||
ctx.isWatched = false
|
||||
close(ctx.watchingStopped)
|
||||
}
|
||||
|
||||
func (ctx *Context) WaitForWatchingStopped(waitCtx context.Context) error {
|
||||
stopped := ctx.watchingStoppedSignal()
|
||||
if stopped == nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
select {
|
||||
case <-waitCtx.Done():
|
||||
return waitCtx.Err()
|
||||
case <-stopped:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
func (ctx *Context) watchingStoppedSignal() <-chan struct{} {
|
||||
ctx.stateMutex.Lock()
|
||||
defer ctx.stateMutex.Unlock()
|
||||
if !ctx.isWatched {
|
||||
return nil
|
||||
}
|
||||
return ctx.watchingStopped
|
||||
}
|
||||
|
||||
34
pkg/servicecontext/servicecontext_race_test.go
Normal file
34
pkg/servicecontext/servicecontext_race_test.go
Normal file
@@ -0,0 +1,34 @@
|
||||
package servicecontext
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestReadinessResetConcurrentWithSignal(t *testing.T) {
|
||||
ctx := New(context.Background())
|
||||
start := make(chan struct{})
|
||||
var wg sync.WaitGroup
|
||||
|
||||
wg.Go(func() {
|
||||
<-start
|
||||
for range 1000 {
|
||||
ctx.SignalReadiness()
|
||||
resetReadiness(ctx)
|
||||
}
|
||||
})
|
||||
wg.Go(func() {
|
||||
<-start
|
||||
for range 1000 {
|
||||
_, ready, _, _ := ctx.ReadinessState()
|
||||
select {
|
||||
case <-ready:
|
||||
default:
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
close(start)
|
||||
wg.Wait()
|
||||
}
|
||||
210
pkg/servicecontext/servicecontext_test.go
Normal file
210
pkg/servicecontext/servicecontext_test.go
Normal file
@@ -0,0 +1,210 @@
|
||||
package servicecontext
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func resetReadiness(ctx *Context) bool {
|
||||
generation, _, _, ready := ctx.ReadinessState()
|
||||
return ready && ctx.ResetReadinessGeneration(generation)
|
||||
}
|
||||
|
||||
func TestReadinessResetCreatesNewGeneration(t *testing.T) {
|
||||
svcCtx := New(context.Background())
|
||||
defer svcCtx.Cancel()
|
||||
|
||||
firstGeneration, first, firstLost, firstReady := svcCtx.ReadinessState()
|
||||
if firstGeneration != 1 || firstReady {
|
||||
t.Fatalf("initial readiness state = generation %d, ready %t; want generation 1, ready false", firstGeneration, firstReady)
|
||||
}
|
||||
svcCtx.SignalReadiness()
|
||||
select {
|
||||
case <-first:
|
||||
default:
|
||||
t.Fatal("first readiness generation was not signalled")
|
||||
}
|
||||
|
||||
if !resetReadiness(svcCtx) {
|
||||
t.Fatal("first readiness generation was not reset")
|
||||
}
|
||||
select {
|
||||
case <-firstLost:
|
||||
default:
|
||||
t.Fatal("first readiness generation loss was not signalled")
|
||||
}
|
||||
|
||||
secondGeneration, second, secondLost, secondReady := svcCtx.ReadinessState()
|
||||
if secondGeneration != firstGeneration+1 || secondReady {
|
||||
t.Fatalf("reset readiness state = generation %d, ready %t; want generation %d, ready false", secondGeneration, secondReady, firstGeneration+1)
|
||||
}
|
||||
if first == second {
|
||||
t.Fatal("readiness reset reused the previous generation")
|
||||
}
|
||||
if firstLost == secondLost {
|
||||
t.Fatal("readiness reset reused the previous loss signal")
|
||||
}
|
||||
select {
|
||||
case <-second:
|
||||
t.Fatal("new readiness generation was already signalled")
|
||||
default:
|
||||
}
|
||||
|
||||
svcCtx.SignalReadiness()
|
||||
select {
|
||||
case <-second:
|
||||
default:
|
||||
t.Fatal("second readiness generation was not signalled")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParentSurvivesServiceCancellation(t *testing.T) {
|
||||
parent, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
svcCtx := New(parent)
|
||||
svcCtx.Cancel()
|
||||
if parent.Err() != nil {
|
||||
t.Fatal("cancelling a service context cancelled its parent")
|
||||
}
|
||||
}
|
||||
|
||||
func TestResetReadinessGenerationWaitsForActivation(t *testing.T) {
|
||||
svcCtx := New(context.Background())
|
||||
defer svcCtx.Cancel()
|
||||
svcCtx.SignalReadiness()
|
||||
generation, _, _, _ := svcCtx.ReadinessState()
|
||||
release, acquired := svcCtx.WaitForReadiness()
|
||||
if !acquired {
|
||||
t.Fatal("ready generation was not available for activation")
|
||||
}
|
||||
|
||||
resetComplete := make(chan bool, 1)
|
||||
go func() {
|
||||
resetComplete <- svcCtx.ResetReadinessGeneration(generation)
|
||||
}()
|
||||
|
||||
deadline := time.Now().Add(time.Second)
|
||||
for {
|
||||
currentGeneration, _, _, _ := svcCtx.ReadinessState()
|
||||
if currentGeneration != generation {
|
||||
break
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatal("readiness reset did not start")
|
||||
}
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
select {
|
||||
case <-resetComplete:
|
||||
t.Fatal("readiness reset returned before the activation released its generation")
|
||||
default:
|
||||
}
|
||||
|
||||
release()
|
||||
select {
|
||||
case reset := <-resetComplete:
|
||||
if !reset {
|
||||
t.Fatal("readiness reset rejected its current generation")
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("readiness reset did not complete after activation released its generation")
|
||||
}
|
||||
}
|
||||
|
||||
func TestStartWatchingClaimsOnce(t *testing.T) {
|
||||
svcCtx := New(context.Background())
|
||||
defer svcCtx.Cancel()
|
||||
|
||||
var claims atomic.Int64
|
||||
var wg sync.WaitGroup
|
||||
for range 32 {
|
||||
wg.Go(func() {
|
||||
if svcCtx.StartWatching() {
|
||||
claims.Add(1)
|
||||
}
|
||||
})
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
if got := claims.Load(); got != 1 {
|
||||
t.Fatalf("watcher claims = %d, want 1", got)
|
||||
}
|
||||
svcCtx.StopWatching()
|
||||
if !svcCtx.StartWatching() {
|
||||
t.Fatal("watcher ownership was not released")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCancelledContextCannotStartWatching(t *testing.T) {
|
||||
svcCtx := New(context.Background())
|
||||
svcCtx.Cancel()
|
||||
if svcCtx.StartWatching() {
|
||||
t.Fatal("cancelled Service context acquired watcher ownership")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWaitForWatchingStopped(t *testing.T) {
|
||||
svcCtx := New(context.Background())
|
||||
defer svcCtx.Cancel()
|
||||
if !svcCtx.StartWatching() {
|
||||
t.Fatal("watcher ownership was not acquired")
|
||||
}
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
done <- svcCtx.WaitForWatchingStopped(context.Background())
|
||||
}()
|
||||
select {
|
||||
case <-done:
|
||||
t.Fatal("WaitForWatchingStopped returned while the watcher was active")
|
||||
case <-time.After(20 * time.Millisecond):
|
||||
}
|
||||
|
||||
svcCtx.StopWatching()
|
||||
select {
|
||||
case err := <-done:
|
||||
if err != nil {
|
||||
t.Fatalf("WaitForWatchingStopped() error = %v", err)
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("WaitForWatchingStopped did not return after watcher shutdown")
|
||||
}
|
||||
}
|
||||
|
||||
func TestConcurrentStateTransitions(t *testing.T) {
|
||||
svcCtx := New(context.Background())
|
||||
defer svcCtx.Cancel()
|
||||
|
||||
var wg sync.WaitGroup
|
||||
for range 100 {
|
||||
wg.Go(func() {
|
||||
svcCtx.SignalReadiness()
|
||||
resetReadiness(svcCtx)
|
||||
})
|
||||
}
|
||||
wg.Wait()
|
||||
if svcCtx.IsReady() && !resetReadiness(svcCtx) {
|
||||
t.Fatal("final readiness generation was not reset")
|
||||
}
|
||||
|
||||
generation, ready, lost, isReady := svcCtx.ReadinessState()
|
||||
if isReady {
|
||||
t.Fatal("concurrent transitions left the context ready after every signal was reset")
|
||||
}
|
||||
if generation == 1 {
|
||||
t.Fatal("concurrent transitions did not advance the readiness generation")
|
||||
}
|
||||
select {
|
||||
case <-ready:
|
||||
t.Fatal("current readiness channel was already closed")
|
||||
default:
|
||||
}
|
||||
select {
|
||||
case <-lost:
|
||||
t.Fatal("current readiness-loss channel was already closed")
|
||||
default:
|
||||
}
|
||||
}
|
||||
@@ -20,5 +20,8 @@ func NewCallback(f func(*servicecontext.Context, *v1.Service, *sync.WaitGroup, b
|
||||
}
|
||||
|
||||
func (c *Callback) Run(svcCtx *servicecontext.Context, svc *v1.Service, wg *sync.WaitGroup) error {
|
||||
if c == nil || c.Function == nil {
|
||||
return nil
|
||||
}
|
||||
return c.Function(svcCtx, svc, wg, c.UsesLeaderElection)
|
||||
}
|
||||
|
||||
31
pkg/services/callback_test.go
Normal file
31
pkg/services/callback_test.go
Normal file
@@ -0,0 +1,31 @@
|
||||
package services
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/servicecontext"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
)
|
||||
|
||||
func TestCallbackRunWithoutFunction(t *testing.T) {
|
||||
for _, callback := range []*Callback{nil, {}} {
|
||||
if err := callback.Run(nil, nil, nil); err != nil {
|
||||
t.Fatalf("Run error = %v", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestCallbackRunPropagatesLeaderElectionFlag(t *testing.T) {
|
||||
want := errors.New("callback error")
|
||||
callback := NewCallback(func(_ *servicecontext.Context, _ *v1.Service, _ *sync.WaitGroup, usesLeaderElection bool) error {
|
||||
if !usesLeaderElection {
|
||||
t.Fatal("callback did not receive leader election flag")
|
||||
}
|
||||
return want
|
||||
}, true)
|
||||
if err := callback.Run(nil, nil, nil); !errors.Is(err, want) {
|
||||
t.Fatalf("Run error = %v, want %v", err, want)
|
||||
}
|
||||
}
|
||||
@@ -7,6 +7,7 @@ import (
|
||||
"fmt"
|
||||
"net"
|
||||
"os"
|
||||
"slices"
|
||||
"strings"
|
||||
|
||||
log "log/slog"
|
||||
@@ -17,6 +18,7 @@ import (
|
||||
"github.com/kube-vip/kube-vip/pkg/utils"
|
||||
"github.com/kube-vip/kube-vip/pkg/vip"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/client-go/util/retry"
|
||||
)
|
||||
@@ -130,6 +132,22 @@ func getSameFamilyCidr(sourceCidrs, ip string) string { //Todo: not sure how thi
|
||||
return cidrs[0]
|
||||
}
|
||||
|
||||
// getSameFamilyCidrs returns every CIDR in the comma-separated list that has the
|
||||
// same IP family as ip.
|
||||
func getSameFamilyCidrs(sourceCidrs, ip string) []string {
|
||||
if sourceCidrs == "" {
|
||||
return nil
|
||||
}
|
||||
isV6 := utils.IsIPv6(ip)
|
||||
var matching []string
|
||||
for _, cidr := range strings.Split(sourceCidrs, ",") {
|
||||
if (isV6 && utils.IsIPv6CIDR(cidr)) || (!isV6 && utils.IsIPv4CIDR(cidr)) {
|
||||
matching = append(matching, cidr)
|
||||
}
|
||||
}
|
||||
return matching
|
||||
}
|
||||
|
||||
func checkCIDR(ip, cidr string) (string, error) {
|
||||
_, ipnetA, err := net.ParseCIDR(cidr)
|
||||
if err != nil {
|
||||
@@ -187,12 +205,12 @@ func (p *Processor) configureEgress(ctx context.Context, vipIP, podIP, namespace
|
||||
log.Warn("autodiscover CIDR", "err", discoverErr)
|
||||
}
|
||||
|
||||
if p.config.EgressPodCidr != "" {
|
||||
podCidr = getSameFamilyCidr(p.config.EgressPodCidr, podIP)
|
||||
} else {
|
||||
if discoverErr == nil {
|
||||
podCidr = getSameFamilyCidr(autoPodCIDR, podIP)
|
||||
}
|
||||
podCidrSource := p.config.EgressPodCidr
|
||||
if podCidrSource == "" && discoverErr == nil {
|
||||
podCidrSource = autoPodCIDR
|
||||
}
|
||||
if podCidrSource != "" {
|
||||
podCidr = getSameFamilyCidr(podCidrSource, podIP)
|
||||
}
|
||||
|
||||
if podCidr == "" {
|
||||
@@ -204,6 +222,14 @@ func (p *Processor) configureEgress(ctx context.Context, vipIP, podIP, namespace
|
||||
podCidr = defaultPodCIDR
|
||||
}
|
||||
|
||||
// Pod-to-pod traffic to any pod CIDR must not be SNAT'd. Node auto-discovery
|
||||
// yields one CIDR per node, so collect every same-family CIDR instead of only
|
||||
// the one containing the local pod IP.
|
||||
podCidrs := getSameFamilyCidrs(podCidrSource, podIP)
|
||||
if len(podCidrs) == 0 {
|
||||
podCidrs = []string{podCidr}
|
||||
}
|
||||
|
||||
if p.config.EgressServiceCidr != "" {
|
||||
serviceCidr = getSameFamilyCidr(p.config.EgressServiceCidr, vipIP)
|
||||
} else {
|
||||
@@ -243,10 +269,8 @@ func (p *Processor) configureEgress(ctx context.Context, vipIP, podIP, namespace
|
||||
var ignoreCIDRs []string
|
||||
// Create an array of CIDRs that we wont SNAT to.
|
||||
if noInternalTraffic == "" || strings.Clone(noInternalTraffic) == "false" || p.config.EnableInternalSNAT {
|
||||
ignoreCIDRs = append(ignoreCIDRs, []string{
|
||||
podCidr,
|
||||
serviceCidr,
|
||||
}...)
|
||||
ignoreCIDRs = append(ignoreCIDRs, podCidrs...)
|
||||
ignoreCIDRs = append(ignoreCIDRs, serviceCidr)
|
||||
}
|
||||
|
||||
// Add any specifically denied networks
|
||||
@@ -300,9 +324,11 @@ func (p *Processor) configureEgress(ctx context.Context, vipIP, podIP, namespace
|
||||
return fmt.Errorf("error creating mangle chain [%s], error [%s]", vip.MangleChainName, err)
|
||||
}
|
||||
}
|
||||
err = i.AppendReturnRulesForDestinationSubnet(vip.MangleChainName, podCidr)
|
||||
if err != nil {
|
||||
return fmt.Errorf("error adding rules to mangle chain [%s], error [%s]", vip.MangleChainName, err)
|
||||
for _, cidr := range podCidrs {
|
||||
err = i.AppendReturnRulesForDestinationSubnet(vip.MangleChainName, cidr)
|
||||
if err != nil {
|
||||
return fmt.Errorf("error adding rules to mangle chain [%s], error [%s]", vip.MangleChainName, err)
|
||||
}
|
||||
}
|
||||
|
||||
err = i.AppendReturnRulesForDestinationSubnet(vip.MangleChainName, serviceCidr)
|
||||
@@ -392,6 +418,49 @@ func (p *Processor) prepareEgressNftablesTable(serviceUID string, ipv6 bool) err
|
||||
|
||||
func (p *Processor) AutoDiscoverCIDRs(ctx context.Context) (serviceCIDR, podCIDR string, err error) {
|
||||
log.Debug("Trying to automatically discover Service and Pod CIDRs")
|
||||
serviceCIDR, podCIDR = p.discoverCIDRsFromAPI(ctx)
|
||||
if serviceCIDR != "" && podCIDR != "" {
|
||||
return serviceCIDR, podCIDR, nil
|
||||
}
|
||||
|
||||
// Fall back to the kube-controller-manager flags; e.g. CNIs doing their own
|
||||
// IPAM don't allocate Node PodCIDRs and clusters may lack the ServiceCIDR API.
|
||||
legacyServiceCIDR, legacyPodCIDR, legacyErr := p.discoverCIDRsFromControllerManager(ctx)
|
||||
if serviceCIDR == "" {
|
||||
serviceCIDR = legacyServiceCIDR
|
||||
}
|
||||
if podCIDR == "" {
|
||||
podCIDR = legacyPodCIDR
|
||||
}
|
||||
if serviceCIDR == "" || podCIDR == "" {
|
||||
if legacyErr != nil {
|
||||
return serviceCIDR, podCIDR, legacyErr
|
||||
}
|
||||
return serviceCIDR, podCIDR, fmt.Errorf("unable to fully determine cluster CIDR configurations")
|
||||
}
|
||||
|
||||
return serviceCIDR, podCIDR, nil
|
||||
}
|
||||
|
||||
func (p *Processor) discoverCIDRsFromAPI(ctx context.Context) (serviceCIDR, podCIDR string) {
|
||||
serviceCIDRs, err := p.clientSet.NetworkingV1().ServiceCIDRs().List(ctx, metav1.ListOptions{})
|
||||
if err == nil {
|
||||
serviceCIDR = serviceCIDRsFromItems(serviceCIDRs.Items)
|
||||
} else {
|
||||
log.Debug("Unable to discover Service CIDRs from ServiceCIDR API", "err", err)
|
||||
}
|
||||
|
||||
nodes, err := p.clientSet.CoreV1().Nodes().List(ctx, metav1.ListOptions{})
|
||||
if err == nil {
|
||||
podCIDR = podCIDRsFromNodes(nodes.Items)
|
||||
} else {
|
||||
log.Debug("Unable to discover CNI Pod CIDRs from Node API", "err", err)
|
||||
}
|
||||
|
||||
return serviceCIDR, podCIDR
|
||||
}
|
||||
|
||||
func (p *Processor) discoverCIDRsFromControllerManager(ctx context.Context) (serviceCIDR, podCIDR string, err error) {
|
||||
options := metav1.ListOptions{
|
||||
LabelSelector: "component=kube-controller-manager",
|
||||
}
|
||||
@@ -404,19 +473,45 @@ func (p *Processor) AutoDiscoverCIDRs(ctx context.Context) (serviceCIDR, podCIDR
|
||||
}
|
||||
|
||||
pod := podList.Items[0]
|
||||
for flags := range pod.Spec.Containers[0].Command {
|
||||
if strings.Contains(pod.Spec.Containers[0].Command[flags], "--cluster-cidr=") {
|
||||
podCIDR = strings.ReplaceAll(pod.Spec.Containers[0].Command[flags], "--cluster-cidr=", "")
|
||||
for _, flag := range pod.Spec.Containers[0].Command {
|
||||
if strings.Contains(flag, "--cluster-cidr=") {
|
||||
podCIDR = strings.ReplaceAll(flag, "--cluster-cidr=", "")
|
||||
}
|
||||
if strings.Contains(pod.Spec.Containers[0].Command[flags], "--service-cluster-ip-range=") {
|
||||
serviceCIDR = strings.ReplaceAll(pod.Spec.Containers[0].Command[flags], "--service-cluster-ip-range=", "")
|
||||
if strings.Contains(flag, "--service-cluster-ip-range=") {
|
||||
serviceCIDR = strings.ReplaceAll(flag, "--service-cluster-ip-range=", "")
|
||||
}
|
||||
}
|
||||
if podCIDR == "" || serviceCIDR == "" {
|
||||
err = fmt.Errorf("unable to fully determine cluster CIDR configurations")
|
||||
}
|
||||
return serviceCIDR, podCIDR, nil
|
||||
}
|
||||
|
||||
return
|
||||
func serviceCIDRsFromItems(items []networkingv1.ServiceCIDR) string {
|
||||
var cidrs []string
|
||||
for _, item := range items {
|
||||
cidrs = appendUnique(cidrs, item.Spec.CIDRs...)
|
||||
}
|
||||
return strings.Join(cidrs, ",")
|
||||
}
|
||||
|
||||
func podCIDRsFromNodes(nodes []corev1.Node) string {
|
||||
var cidrs []string
|
||||
for _, node := range nodes {
|
||||
if len(node.Spec.PodCIDRs) > 0 {
|
||||
cidrs = appendUnique(cidrs, node.Spec.PodCIDRs...)
|
||||
} else if node.Spec.PodCIDR != "" {
|
||||
cidrs = appendUnique(cidrs, node.Spec.PodCIDR)
|
||||
}
|
||||
}
|
||||
return strings.Join(cidrs, ",")
|
||||
}
|
||||
|
||||
func appendUnique(values []string, additions ...string) []string {
|
||||
for _, addition := range additions {
|
||||
if addition == "" || slices.Contains(values, addition) {
|
||||
continue
|
||||
}
|
||||
values = append(values, addition)
|
||||
}
|
||||
return values
|
||||
}
|
||||
|
||||
func (p *Processor) updateEgressNftablesTableAnnotation(ctx context.Context, service *corev1.Service) error {
|
||||
|
||||
79
pkg/services/egress_cidr_test.go
Normal file
79
pkg/services/egress_cidr_test.go
Normal file
@@ -0,0 +1,79 @@
|
||||
package services
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
)
|
||||
|
||||
func TestServiceCIDRsFromItems(t *testing.T) {
|
||||
serviceCIDR := serviceCIDRsFromItems([]networkingv1.ServiceCIDR{
|
||||
{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "kubernetes"},
|
||||
Spec: networkingv1.ServiceCIDRSpec{
|
||||
CIDRs: []string{"10.96.0.0/16", "fd00:10:96::/112"},
|
||||
},
|
||||
},
|
||||
})
|
||||
|
||||
if serviceCIDR != "10.96.0.0/16,fd00:10:96::/112" {
|
||||
t.Fatalf("serviceCIDR = %q", serviceCIDR)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPodCIDRsFromNodes(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
nodes []corev1.Node
|
||||
want string
|
||||
}{
|
||||
{
|
||||
name: "dual stack across nodes",
|
||||
nodes: []corev1.Node{
|
||||
{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "node-a"},
|
||||
Spec: corev1.NodeSpec{
|
||||
PodCIDRs: []string{"10.244.0.0/24", "fd00:10:244::/64"},
|
||||
},
|
||||
},
|
||||
{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "node-b"},
|
||||
Spec: corev1.NodeSpec{
|
||||
PodCIDRs: []string{"10.244.1.0/24", "fd00:10:244:1::/64"},
|
||||
},
|
||||
},
|
||||
},
|
||||
want: "10.244.0.0/24,fd00:10:244::/64,10.244.1.0/24,fd00:10:244:1::/64",
|
||||
},
|
||||
{
|
||||
name: "legacy singular CIDR",
|
||||
nodes: []corev1.Node{
|
||||
{Spec: corev1.NodeSpec{PodCIDR: "10.244.0.0/24"}},
|
||||
},
|
||||
want: "10.244.0.0/24",
|
||||
},
|
||||
{
|
||||
name: "duplicate CIDRs",
|
||||
nodes: []corev1.Node{
|
||||
{Spec: corev1.NodeSpec{PodCIDRs: []string{"10.244.0.0/16"}}},
|
||||
{Spec: corev1.NodeSpec{PodCIDRs: []string{"10.244.0.0/16"}}},
|
||||
},
|
||||
want: "10.244.0.0/16",
|
||||
},
|
||||
{
|
||||
name: "missing CNI CIDRs",
|
||||
nodes: []corev1.Node{{}},
|
||||
want: "",
|
||||
},
|
||||
}
|
||||
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
if got := podCIDRsFromNodes(test.nodes); got != test.want {
|
||||
t.Fatalf("podCIDRsFromNodes() = %q, want %q", got, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
838
pkg/services/election.go
Normal file
838
pkg/services/election.go
Normal file
@@ -0,0 +1,838 @@
|
||||
package services
|
||||
|
||||
import (
|
||||
"context"
|
||||
log "log/slog"
|
||||
"sort"
|
||||
"strconv"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/election"
|
||||
"github.com/kube-vip/kube-vip/pkg/instance"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
"github.com/kube-vip/kube-vip/pkg/metrics"
|
||||
"github.com/kube-vip/kube-vip/pkg/servicecontext"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
)
|
||||
|
||||
// serviceElection owns local membership and campaign lifetime for one lease.
|
||||
// Service contexts remain responsible for endpoint readiness and datapath work.
|
||||
type serviceElection struct {
|
||||
processor *Processor
|
||||
id lease.ID
|
||||
|
||||
mutex sync.Mutex
|
||||
members map[types.UID]*serviceElectionMember
|
||||
lease *lease.Lease
|
||||
campaign *serviceElectionCampaign
|
||||
retired bool
|
||||
retiredDone chan struct{}
|
||||
retiredCtx context.Context
|
||||
retireCancel context.CancelFunc
|
||||
|
||||
// restartFailures counts consecutive campaigns that ended via
|
||||
// cancelCampaign (an activation failure with no other ready member)
|
||||
// rather than a normal leadership change. It backs off campaign restarts
|
||||
// and resets on the next successful activation.
|
||||
restartFailures int
|
||||
}
|
||||
|
||||
const (
|
||||
serviceElectionRestartBaseDelay = 200 * time.Millisecond
|
||||
serviceElectionRestartMaxDelay = 30 * time.Second
|
||||
)
|
||||
|
||||
type serviceElectionCampaign struct {
|
||||
done chan struct{}
|
||||
ctx context.Context
|
||||
cancel context.CancelFunc
|
||||
leaderCtx context.Context
|
||||
cancelLeader context.CancelFunc
|
||||
vips []string
|
||||
external bool
|
||||
stopped bool
|
||||
}
|
||||
|
||||
func (campaign *serviceElectionCampaign) cancelRunner() {
|
||||
if campaign != nil && campaign.cancel != nil {
|
||||
campaign.cancel()
|
||||
}
|
||||
}
|
||||
|
||||
type serviceElectionMember struct {
|
||||
election *serviceElection
|
||||
service *v1.Service
|
||||
serviceContext *servicecontext.Context
|
||||
readinessGeneration uint64
|
||||
claimToken string
|
||||
operationMutex sync.Mutex
|
||||
active bool
|
||||
}
|
||||
|
||||
func (p *Processor) serviceElectionFor(id lease.ID) *serviceElection {
|
||||
p.electionsMutex.Lock()
|
||||
defer p.electionsMutex.Unlock()
|
||||
if p.elections == nil {
|
||||
p.elections = make(map[string]*serviceElection)
|
||||
}
|
||||
key := id.NamespacedName()
|
||||
if election := p.elections[key]; election != nil {
|
||||
return election
|
||||
}
|
||||
retiredCtx, retire := context.WithCancel(context.Background())
|
||||
election := &serviceElection{
|
||||
processor: p,
|
||||
id: id,
|
||||
members: make(map[types.UID]*serviceElectionMember),
|
||||
retiredDone: make(chan struct{}),
|
||||
retiredCtx: retiredCtx,
|
||||
retireCancel: retire,
|
||||
}
|
||||
p.elections[key] = election
|
||||
return election
|
||||
}
|
||||
|
||||
// joinServiceElection registers the current ready generation of a Service. A
|
||||
// caller that races coordinator retirement retries against its replacement.
|
||||
func (p *Processor) joinServiceElection(svcCtx *servicecontext.Context, service *v1.Service,
|
||||
readinessGeneration uint64) (*serviceElectionMember, bool) {
|
||||
if svcCtx == nil || service == nil || p.leaseMgr == nil {
|
||||
return nil, false
|
||||
}
|
||||
|
||||
namespace, name := lease.ServiceName(service)
|
||||
id := lease.NewID(p.config.LeaderElectionType, namespace, name)
|
||||
for {
|
||||
if !p.serviceElectionContextCurrent(svcCtx, service, readinessGeneration) {
|
||||
return nil, false
|
||||
}
|
||||
election := p.serviceElectionFor(id)
|
||||
member, joined := election.join(svcCtx, service, readinessGeneration)
|
||||
if joined {
|
||||
if p.serviceElectionContextCurrent(svcCtx, service, readinessGeneration) {
|
||||
return member, true
|
||||
}
|
||||
p.leaveServiceElection(member)
|
||||
return nil, false
|
||||
}
|
||||
if retiredDone, retired := election.retirement(); retired {
|
||||
select {
|
||||
case <-svcCtx.Ctx.Done():
|
||||
return nil, false
|
||||
case <-retiredDone:
|
||||
continue
|
||||
}
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
}
|
||||
|
||||
// serviceElectionContextCurrent acquires the Service lock while comparing the
|
||||
// current context. Callers must not already hold that lock.
|
||||
func (p *Processor) serviceElectionContextCurrent(svcCtx *servicecontext.Context, service *v1.Service, readinessGeneration uint64) bool {
|
||||
currentContext, err := p.currentServiceContext(service.UID)
|
||||
if err != nil || currentContext != svcCtx || svcCtx.Ctx.Err() != nil {
|
||||
return false
|
||||
}
|
||||
return svcCtx.ReadinessGenerationCurrent(readinessGeneration)
|
||||
}
|
||||
|
||||
// currentServiceContext acquires and releases the Service lock around one
|
||||
// svcMap read.
|
||||
func (p *Processor) currentServiceContext(uid types.UID) (*servicecontext.Context, error) {
|
||||
unlockService := p.lockService(uid)
|
||||
defer unlockService()
|
||||
return p.getServiceContext(uid)
|
||||
}
|
||||
|
||||
func (e *serviceElection) join(svcCtx *servicecontext.Context, service *v1.Service,
|
||||
readinessGeneration uint64) (*serviceElectionMember, bool) {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
if e.retired {
|
||||
return nil, false
|
||||
}
|
||||
if member := e.members[service.UID]; member != nil && member.serviceContext == svcCtx &&
|
||||
member.readinessGeneration == readinessGeneration {
|
||||
return member, true
|
||||
}
|
||||
|
||||
if previous := e.members[service.UID]; previous != nil {
|
||||
e.processor.leaseMgr.Delete(e.id, previous.claimToken, e.lease)
|
||||
}
|
||||
member := &serviceElectionMember{
|
||||
election: e,
|
||||
service: service.DeepCopy(),
|
||||
serviceContext: svcCtx,
|
||||
readinessGeneration: readinessGeneration,
|
||||
claimToken: e.nextMemberToken(),
|
||||
}
|
||||
e.members[service.UID] = member
|
||||
|
||||
// A member can become ready again while the old campaign is still stopping.
|
||||
// Keep its new generation until that runner finishes; finishCampaign will
|
||||
// rebuild the lease and launch the replacement campaign.
|
||||
if e.lease != nil && e.lease.Ctx.Err() != nil && e.campaign != nil {
|
||||
return member, true
|
||||
}
|
||||
if e.lease == nil || e.lease.Ctx.Err() != nil {
|
||||
if e.createLeaseLocked() == nil {
|
||||
delete(e.members, service.UID)
|
||||
return nil, false
|
||||
}
|
||||
return member, true
|
||||
}
|
||||
if claimed, _ := e.processor.leaseMgr.Claim(e.id, member.claimToken); claimed != nil {
|
||||
return member, true
|
||||
}
|
||||
|
||||
// An external cleanup retired the manager entry. Rebuild it from the live
|
||||
// coordinator snapshot rather than admitting a member to a dead lease.
|
||||
e.lease = nil
|
||||
if e.createLeaseLocked() == nil {
|
||||
delete(e.members, service.UID)
|
||||
return nil, false
|
||||
}
|
||||
return member, true
|
||||
}
|
||||
|
||||
func (e *serviceElection) createLeaseLocked() *lease.Lease {
|
||||
if e.lease != nil && e.lease.Ctx.Err() == nil {
|
||||
return e.lease
|
||||
}
|
||||
var first *serviceElectionMember
|
||||
for _, member := range e.members {
|
||||
first = member
|
||||
break
|
||||
}
|
||||
if first == nil {
|
||||
return nil
|
||||
}
|
||||
svcLease, _ := e.processor.leaseMgr.Acquire(context.Background(), e.id, first.claimToken)
|
||||
for _, member := range e.members {
|
||||
if member == first {
|
||||
continue
|
||||
}
|
||||
if claimed, _ := e.processor.leaseMgr.Claim(e.id, member.claimToken); claimed == nil {
|
||||
svcLease.Cancel()
|
||||
return nil
|
||||
}
|
||||
}
|
||||
e.lease = svcLease
|
||||
return svcLease
|
||||
}
|
||||
|
||||
func (e *serviceElection) nextMemberToken() string {
|
||||
return strconv.FormatUint(e.processor.nextMemberToken.Add(1), 10)
|
||||
}
|
||||
|
||||
// leaveServiceElection removes only the supplied member generation. A stale
|
||||
// member cannot remove a replacement Service context or readiness generation.
|
||||
func (p *Processor) leaveServiceElection(member *serviceElectionMember) {
|
||||
if member == nil {
|
||||
return
|
||||
}
|
||||
member.election.leave(member)
|
||||
}
|
||||
|
||||
func (p *Processor) leaveServiceElectionForContext(svcCtx *servicecontext.Context, service *v1.Service) {
|
||||
if svcCtx == nil || service == nil {
|
||||
return
|
||||
}
|
||||
namespace, name := lease.ServiceName(service)
|
||||
id := lease.NewID(p.config.LeaderElectionType, namespace, name)
|
||||
election := p.currentServiceElection(id)
|
||||
if election == nil {
|
||||
return
|
||||
}
|
||||
member := election.currentMember(service.UID)
|
||||
if member != nil && member.serviceContext == svcCtx {
|
||||
p.leaveServiceElection(member)
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Processor) currentServiceElection(id lease.ID) *serviceElection {
|
||||
p.electionsMutex.Lock()
|
||||
defer p.electionsMutex.Unlock()
|
||||
return p.elections[id.NamespacedName()]
|
||||
}
|
||||
|
||||
func (e *serviceElection) currentMember(uid types.UID) *serviceElectionMember {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
return e.members[uid]
|
||||
}
|
||||
|
||||
func (e *serviceElection) leave(member *serviceElectionMember) {
|
||||
campaign, leaseRetired, retired := e.removeMember(member)
|
||||
if !retired {
|
||||
return
|
||||
}
|
||||
e.retire()
|
||||
if campaign != nil && (campaign.external || leaseRetired) {
|
||||
campaign.cancelRunner()
|
||||
}
|
||||
// A Service-owned runner remains responsible for the shared election when
|
||||
// a non-Service member still holds the lease. Its lease-scoped context ends
|
||||
// when that final member leaves or the election itself stops.
|
||||
}
|
||||
|
||||
// removeMember deletes member if it is still current and, once no members
|
||||
// remain, marks the election retired and reports whether deleting its claim
|
||||
// also retired the shared lease.
|
||||
func (e *serviceElection) removeMember(member *serviceElectionMember) (campaign *serviceElectionCampaign, leaseRetired, retired bool) {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
if e.members[member.service.UID] != member {
|
||||
return nil, false, false
|
||||
}
|
||||
delete(e.members, member.service.UID)
|
||||
leaseRetired = e.processor.leaseMgr.Delete(e.id, member.claimToken, e.lease)
|
||||
if len(e.members) != 0 {
|
||||
return nil, leaseRetired, false
|
||||
}
|
||||
e.retired = true
|
||||
campaign = e.campaign
|
||||
e.lease = nil
|
||||
return campaign, leaseRetired, true
|
||||
}
|
||||
|
||||
func (e *serviceElection) retirement() (<-chan struct{}, bool) {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
return e.retiredDone, e.retired
|
||||
}
|
||||
|
||||
func (e *serviceElection) retire() {
|
||||
if e.retireCancel != nil {
|
||||
e.retireCancel()
|
||||
}
|
||||
e.processor.removeServiceElection(e)
|
||||
close(e.retiredDone)
|
||||
}
|
||||
|
||||
func (p *Processor) removeServiceElection(election *serviceElection) {
|
||||
p.electionsMutex.Lock()
|
||||
defer p.electionsMutex.Unlock()
|
||||
if p.elections[election.id.NamespacedName()] == election {
|
||||
delete(p.elections, election.id.NamespacedName())
|
||||
}
|
||||
}
|
||||
|
||||
func (p *Processor) watchServiceElection(svcCtx *servicecontext.Context, service *v1.Service,
|
||||
wg *sync.WaitGroup) {
|
||||
for {
|
||||
generation, ready, lost, isReady := svcCtx.ReadinessState()
|
||||
if !isReady {
|
||||
select {
|
||||
case <-svcCtx.Ctx.Done():
|
||||
return
|
||||
case <-ready:
|
||||
continue
|
||||
}
|
||||
}
|
||||
|
||||
member, joined := p.joinServiceElection(svcCtx, service, generation)
|
||||
if !joined {
|
||||
if !p.serviceElectionContextCurrent(svcCtx, service, generation) {
|
||||
return
|
||||
}
|
||||
select {
|
||||
case <-svcCtx.Ctx.Done():
|
||||
return
|
||||
case <-time.After(serviceElectionRestartBaseDelay):
|
||||
continue
|
||||
}
|
||||
}
|
||||
member.election.startCampaign(wg)
|
||||
|
||||
select {
|
||||
case <-svcCtx.Ctx.Done():
|
||||
member.election.deactivateMember(member)
|
||||
p.leaveServiceElection(member)
|
||||
return
|
||||
case <-lost:
|
||||
member.election.deactivateMember(member)
|
||||
p.leaveServiceElection(member)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// campaignStart carries the decision taken under the election mutex so the
|
||||
// caller can act on it without holding the lock.
|
||||
type campaignStart struct {
|
||||
lease *lease.Lease
|
||||
campaign *serviceElectionCampaign
|
||||
leaderCtx context.Context
|
||||
members []*serviceElectionMember
|
||||
joinExisting bool
|
||||
external bool
|
||||
}
|
||||
|
||||
func (e *serviceElection) prepareCampaign() campaignStart {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
|
||||
if e.retired || len(e.members) == 0 {
|
||||
return campaignStart{}
|
||||
}
|
||||
if e.campaign != nil {
|
||||
return campaignStart{
|
||||
lease: e.lease,
|
||||
campaign: e.campaign,
|
||||
leaderCtx: e.campaign.leaderCtx,
|
||||
joinExisting: true,
|
||||
}
|
||||
}
|
||||
svcLease := e.createLeaseLocked()
|
||||
if svcLease == nil {
|
||||
return campaignStart{}
|
||||
}
|
||||
external := !svcLease.BeginElection()
|
||||
campaignCtx, campaignCancel := svcLease.NewElectionContext(context.Background())
|
||||
members := e.membersLocked()
|
||||
campaign := &serviceElectionCampaign{
|
||||
done: make(chan struct{}),
|
||||
ctx: campaignCtx,
|
||||
cancel: campaignCancel,
|
||||
vips: memberVIPs(members),
|
||||
external: external,
|
||||
}
|
||||
e.campaign = campaign
|
||||
return campaignStart{lease: svcLease, campaign: campaign, members: members, external: external}
|
||||
}
|
||||
|
||||
func (e *serviceElection) startCampaign(wg *sync.WaitGroup) {
|
||||
start := e.prepareCampaign()
|
||||
if start.campaign == nil {
|
||||
return
|
||||
}
|
||||
if start.joinExisting {
|
||||
if start.leaderCtx != nil {
|
||||
e.activateMembers(start.leaderCtx, start.lease, start.campaign, wg)
|
||||
}
|
||||
return
|
||||
}
|
||||
if start.external {
|
||||
wg.Go(func() {
|
||||
e.followCampaign(start.lease, start.campaign, wg)
|
||||
})
|
||||
return
|
||||
}
|
||||
for _, member := range start.members {
|
||||
metrics.ServiceElectionAttemptsTotal.WithLabelValues(member.service.Namespace, member.service.Name).Inc()
|
||||
}
|
||||
|
||||
wg.Go(func() {
|
||||
e.runCampaign(start.lease, start.campaign, wg)
|
||||
})
|
||||
}
|
||||
|
||||
// adoptLeaderContext publishes the leader context for a campaign that just won
|
||||
// an externally driven election.
|
||||
func (e *serviceElection) adoptLeaderContext(svcLease *lease.Lease, campaign *serviceElectionCampaign,
|
||||
leaderCtx context.Context, cancelLeader context.CancelFunc) bool {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
|
||||
if e.retired || e.lease != svcLease || e.campaign != campaign || campaign.stopped {
|
||||
return false
|
||||
}
|
||||
campaign.leaderCtx = leaderCtx
|
||||
campaign.cancelLeader = cancelLeader
|
||||
return true
|
||||
}
|
||||
|
||||
func (e *serviceElection) followCampaign(svcLease *lease.Lease, campaign *serviceElectionCampaign, wg *sync.WaitGroup) {
|
||||
defer close(campaign.done)
|
||||
defer campaign.cancelRunner()
|
||||
leaderGeneration, elected := svcLease.WaitForLeaderGeneration(campaign.ctx)
|
||||
if !elected {
|
||||
e.stopCampaign(svcLease, campaign)
|
||||
e.finishCampaign(svcLease, campaign, wg)
|
||||
return
|
||||
}
|
||||
leaderCtx, cancelLeader := context.WithCancel(campaign.ctx)
|
||||
if !e.adoptLeaderContext(svcLease, campaign, leaderCtx, cancelLeader) {
|
||||
cancelLeader()
|
||||
return
|
||||
}
|
||||
e.activateMembers(leaderCtx, svcLease, campaign, wg)
|
||||
svcLease.WaitForElectionEndAfter(campaign.ctx, leaderGeneration)
|
||||
cancelLeader()
|
||||
e.stopCampaign(svcLease, campaign)
|
||||
e.finishCampaign(svcLease, campaign, wg)
|
||||
}
|
||||
|
||||
func (e *serviceElection) runCampaign(svcLease *lease.Lease, campaign *serviceElectionCampaign, wg *sync.WaitGroup) {
|
||||
defer close(campaign.done)
|
||||
defer campaign.cancelRunner()
|
||||
run := election.RunConfig{
|
||||
Config: e.processor.config,
|
||||
LeaseID: e.id,
|
||||
Mgr: e.processor.electionMgr,
|
||||
LeaseAnnotations: map[string]string{},
|
||||
VIPs: campaign.vips,
|
||||
OnStartedLeading: func(ctx context.Context) {
|
||||
e.startedLeading(ctx, svcLease, campaign, wg)
|
||||
},
|
||||
OnStoppedLeading: func() {
|
||||
e.stopCampaign(svcLease, campaign)
|
||||
metrics.IsLeader.WithLabelValues(e.processor.config.NodeName, e.id.Name()).Set(0)
|
||||
},
|
||||
OnNewLeader: func(identity string) {
|
||||
if identity != e.processor.config.NodeName {
|
||||
log.Info("new leader", "leader", identity, "lease", e.id.NamespacedName())
|
||||
}
|
||||
},
|
||||
}
|
||||
if err := e.processor.runElection(campaign.ctx, &run); err != nil {
|
||||
log.Error("services election failed", "lease", e.id.NamespacedName(), "error", err)
|
||||
}
|
||||
e.stopCampaign(svcLease, campaign)
|
||||
svcLease.ElectionStopped()
|
||||
e.finishCampaign(svcLease, campaign, wg)
|
||||
}
|
||||
|
||||
func memberVIPs(members []*serviceElectionMember) []string {
|
||||
services := make([]*v1.Service, 0, len(members))
|
||||
for _, member := range members {
|
||||
if member != nil && member.service != nil {
|
||||
services = append(services, member.service)
|
||||
}
|
||||
}
|
||||
return orderedServiceVIPs(services)
|
||||
}
|
||||
|
||||
func orderedServiceVIPs(services []*v1.Service) []string {
|
||||
services = append([]*v1.Service(nil), services...)
|
||||
sort.SliceStable(services, func(first, second int) bool {
|
||||
firstService, secondService := services[first], services[second]
|
||||
if !firstService.CreationTimestamp.Equal(&secondService.CreationTimestamp) {
|
||||
return firstService.CreationTimestamp.Before(&secondService.CreationTimestamp)
|
||||
}
|
||||
if firstService.Namespace != secondService.Namespace {
|
||||
return firstService.Namespace < secondService.Namespace
|
||||
}
|
||||
if firstService.Name != secondService.Name {
|
||||
return firstService.Name < secondService.Name
|
||||
}
|
||||
return firstService.UID < secondService.UID
|
||||
})
|
||||
|
||||
vips := make([]string, 0)
|
||||
for _, service := range services {
|
||||
addresses, _ := instance.FetchServiceAddresses(service)
|
||||
vips = append(vips, addresses...)
|
||||
}
|
||||
return vips
|
||||
}
|
||||
|
||||
func (p *Processor) runElection(ctx context.Context, run *election.RunConfig) error {
|
||||
if p.electionRun != nil {
|
||||
return p.electionRun(ctx, run, p.config)
|
||||
}
|
||||
return election.RunOrDie(ctx, run, p.config)
|
||||
}
|
||||
|
||||
func (e *serviceElection) membersLocked() []*serviceElectionMember {
|
||||
members := make([]*serviceElectionMember, 0, len(e.members))
|
||||
for _, member := range e.members {
|
||||
members = append(members, member)
|
||||
}
|
||||
return members
|
||||
}
|
||||
|
||||
// beginLeading records the leader context for a campaign this process won.
|
||||
func (e *serviceElection) beginLeading(ctx context.Context, svcLease *lease.Lease,
|
||||
campaign *serviceElectionCampaign) bool {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
|
||||
if e.retired || e.lease != svcLease || e.campaign != campaign || campaign.stopped || len(e.members) == 0 {
|
||||
return false
|
||||
}
|
||||
campaign.leaderCtx = ctx
|
||||
svcLease.ElectionStarted()
|
||||
return true
|
||||
}
|
||||
|
||||
func (e *serviceElection) startedLeading(ctx context.Context, svcLease *lease.Lease,
|
||||
campaign *serviceElectionCampaign, wg *sync.WaitGroup) {
|
||||
if !e.beginLeading(ctx, svcLease, campaign) {
|
||||
return
|
||||
}
|
||||
metrics.LeaderTransitionsTotal.WithLabelValues(e.id.Name()).Inc()
|
||||
metrics.IsLeader.WithLabelValues(e.processor.config.NodeName, e.id.Name()).Set(1)
|
||||
e.activateMembers(ctx, svcLease, campaign, wg)
|
||||
}
|
||||
|
||||
// activatableMembers snapshots the members eligible for activation, or nil when
|
||||
// the campaign is no longer current.
|
||||
func (e *serviceElection) activatableMembers(svcLease *lease.Lease,
|
||||
campaign *serviceElectionCampaign) []*serviceElectionMember {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
|
||||
if e.retired || e.lease != svcLease || e.campaign != campaign || (campaign != nil && campaign.stopped) || !svcLease.Elected.Load() {
|
||||
return nil
|
||||
}
|
||||
return e.membersLocked()
|
||||
}
|
||||
|
||||
func (e *serviceElection) activateMembers(ctx context.Context, svcLease *lease.Lease,
|
||||
campaign *serviceElectionCampaign, wg *sync.WaitGroup) {
|
||||
if svcLease == nil || !svcLease.Elected.Load() {
|
||||
return
|
||||
}
|
||||
for _, member := range e.activatableMembers(svcLease, campaign) {
|
||||
e.activateMember(ctx, member, svcLease, campaign, wg)
|
||||
}
|
||||
}
|
||||
|
||||
func (e *serviceElection) activateMember(ctx context.Context, member *serviceElectionMember, svcLease *lease.Lease,
|
||||
campaign *serviceElectionCampaign, wg *sync.WaitGroup) {
|
||||
releaseReadiness, ready := member.serviceContext.AcquireReadinessGeneration(member.readinessGeneration)
|
||||
if !ready {
|
||||
return
|
||||
}
|
||||
defer releaseReadiness()
|
||||
|
||||
member.operationMutex.Lock()
|
||||
defer member.operationMutex.Unlock()
|
||||
|
||||
if !e.processor.serviceElectionMemberCurrent(member) || !e.markMemberActive(member, svcLease, campaign) {
|
||||
return
|
||||
}
|
||||
if err := e.processor.syncServices(ctx, member.serviceContext, member.service, wg, true); err != nil {
|
||||
metrics.ServiceElectionErrorsTotal.WithLabelValues(member.service.Namespace, member.service.Name, "service_sync").Inc()
|
||||
log.Error("start service after election", "service", member.service.Name, "namespace", member.service.Namespace, "error", err)
|
||||
e.deactivateMemberOperationHeld(member)
|
||||
if !e.hasOtherReadyMember(member) {
|
||||
e.cancelCampaign(svcLease, campaign)
|
||||
}
|
||||
return
|
||||
}
|
||||
e.resetRestartFailures()
|
||||
if !e.processor.serviceElectionMemberCurrent(member) || !e.memberActivationCurrent(member, svcLease, campaign) {
|
||||
e.deactivateMemberOperationHeld(member)
|
||||
}
|
||||
}
|
||||
|
||||
func (e *serviceElection) resetRestartFailures() {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
e.restartFailures = 0
|
||||
}
|
||||
|
||||
func (p *Processor) syncServices(operationCtx context.Context, svcCtx *servicecontext.Context,
|
||||
service *v1.Service, wg *sync.WaitGroup, usesLeaderElection bool) error {
|
||||
if p.serviceSync != nil {
|
||||
return p.serviceSync(operationCtx, svcCtx, service, wg, usesLeaderElection)
|
||||
}
|
||||
return p.syncServicesWithContext(operationCtx, svcCtx, service, wg, usesLeaderElection)
|
||||
}
|
||||
|
||||
func (e *serviceElection) memberActivationCurrent(member *serviceElectionMember, svcLease *lease.Lease,
|
||||
campaign *serviceElectionCampaign) bool {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
return e.memberActivationCurrentLocked(member, svcLease, campaign) && member.active
|
||||
}
|
||||
|
||||
func (e *serviceElection) memberActivationCurrentLocked(member *serviceElectionMember, svcLease *lease.Lease,
|
||||
campaign *serviceElectionCampaign) bool {
|
||||
return !e.retired && e.lease == svcLease && e.campaign == campaign &&
|
||||
(campaign == nil || !campaign.stopped) && svcLease.Elected.Load() &&
|
||||
e.members[member.service.UID] == member
|
||||
}
|
||||
|
||||
func (e *serviceElection) markMemberActive(member *serviceElectionMember, svcLease *lease.Lease, campaign *serviceElectionCampaign) bool {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
if !e.memberActivationCurrentLocked(member, svcLease, campaign) || member.active {
|
||||
return false
|
||||
}
|
||||
member.active = true
|
||||
return true
|
||||
}
|
||||
|
||||
func (e *serviceElection) hasOtherReadyMember(member *serviceElectionMember) bool {
|
||||
for _, candidate := range e.otherMembers(member) {
|
||||
if candidate.serviceContext.Ctx.Err() == nil && candidate.serviceContext.ReadinessGenerationCurrent(candidate.readinessGeneration) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// otherMembers returns every member except the supplied one.
|
||||
func (e *serviceElection) otherMembers(member *serviceElectionMember) []*serviceElectionMember {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
others := make([]*serviceElectionMember, 0, len(e.members))
|
||||
for _, candidate := range e.members {
|
||||
if candidate != member {
|
||||
others = append(others, candidate)
|
||||
}
|
||||
}
|
||||
return others
|
||||
}
|
||||
|
||||
func (e *serviceElection) deactivateMember(member *serviceElectionMember) {
|
||||
member.operationMutex.Lock()
|
||||
defer member.operationMutex.Unlock()
|
||||
e.deactivateMemberOperationHeld(member)
|
||||
}
|
||||
|
||||
// markMemberInactive clears the active flag and reports the lease that the
|
||||
// caller must run cleanup against.
|
||||
func (e *serviceElection) markMemberInactive(member *serviceElectionMember) (*lease.Lease, bool) {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
|
||||
if e.members[member.service.UID] != member || !member.active {
|
||||
return nil, false
|
||||
}
|
||||
member.active = false
|
||||
return e.lease, true
|
||||
}
|
||||
|
||||
func (e *serviceElection) deactivateMemberOperationHeld(member *serviceElectionMember) {
|
||||
svcLease, deactivated := e.markMemberInactive(member)
|
||||
if !deactivated {
|
||||
return
|
||||
}
|
||||
e.cleanupMember(member, svcLease)
|
||||
}
|
||||
|
||||
// serviceElectionMemberCurrent acquires and releases the Service lock before
|
||||
// acquiring the election mutex. Callers must not already hold the Service lock.
|
||||
func (p *Processor) serviceElectionMemberCurrent(member *serviceElectionMember) bool {
|
||||
currentCtx, err := p.currentServiceContext(member.service.UID)
|
||||
current := err == nil && currentCtx == member.serviceContext && member.serviceContext.Ctx.Err() == nil &&
|
||||
member.serviceContext.ReadinessGenerationCurrent(member.readinessGeneration)
|
||||
if !current {
|
||||
return false
|
||||
}
|
||||
|
||||
member.election.mutex.Lock()
|
||||
defer member.election.mutex.Unlock()
|
||||
return !member.election.retired && member.election.members[member.service.UID] == member
|
||||
}
|
||||
|
||||
func (e *serviceElection) cleanupMember(member *serviceElectionMember, svcLease *lease.Lease) {
|
||||
if svcLease == nil {
|
||||
return
|
||||
}
|
||||
if err := e.processor.onStoppedLeadingMember(member, svcLease); err != nil {
|
||||
log.Error("stop service after election", "service", member.service.Name, "namespace", member.service.Namespace, "error", err)
|
||||
}
|
||||
}
|
||||
|
||||
// markCampaignStopped retires the campaign and returns the members whose
|
||||
// datapath the caller must tear down outside the lock.
|
||||
func (e *serviceElection) markCampaignStopped(svcLease *lease.Lease,
|
||||
campaign *serviceElectionCampaign) []*serviceElectionMember {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
|
||||
if e.retired || e.lease != svcLease || e.campaign != campaign || campaign.stopped {
|
||||
return nil
|
||||
}
|
||||
campaign.stopped = true
|
||||
if campaign.cancelLeader != nil {
|
||||
campaign.cancelLeader()
|
||||
}
|
||||
if !campaign.external {
|
||||
svcLease.ElectionStopped()
|
||||
}
|
||||
return e.membersLocked()
|
||||
}
|
||||
|
||||
func (e *serviceElection) stopCampaign(svcLease *lease.Lease, campaign *serviceElectionCampaign) {
|
||||
for _, member := range e.markCampaignStopped(svcLease, campaign) {
|
||||
e.deactivateMember(member)
|
||||
}
|
||||
}
|
||||
|
||||
// recordCampaignFailure counts an activation failure for the restart backoff.
|
||||
func (e *serviceElection) recordCampaignFailure(svcLease *lease.Lease, campaign *serviceElectionCampaign) bool {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
|
||||
if e.retired || e.lease != svcLease || e.campaign != campaign {
|
||||
return false
|
||||
}
|
||||
e.restartFailures++
|
||||
return true
|
||||
}
|
||||
|
||||
func (e *serviceElection) cancelCampaign(svcLease *lease.Lease, campaign *serviceElectionCampaign) {
|
||||
if !e.recordCampaignFailure(svcLease, campaign) {
|
||||
return
|
||||
}
|
||||
campaign.cancelRunner()
|
||||
}
|
||||
|
||||
// completeCampaign clears the finished campaign and reports whether a restart
|
||||
// is still needed, along with its backoff delay.
|
||||
func (e *serviceElection) completeCampaign(svcLease *lease.Lease,
|
||||
campaign *serviceElectionCampaign) (bool, time.Duration) {
|
||||
e.mutex.Lock()
|
||||
defer e.mutex.Unlock()
|
||||
|
||||
if e.retired || e.lease != svcLease || e.campaign != campaign {
|
||||
return false, 0
|
||||
}
|
||||
e.campaign = nil
|
||||
if svcLease.Ctx.Err() != nil {
|
||||
e.lease = nil
|
||||
}
|
||||
return len(e.members) != 0, e.restartDelayLocked()
|
||||
}
|
||||
|
||||
func (e *serviceElection) finishCampaign(svcLease *lease.Lease, campaign *serviceElectionCampaign, wg *sync.WaitGroup) {
|
||||
restart, delay := e.completeCampaign(svcLease, campaign)
|
||||
if !restart {
|
||||
return
|
||||
}
|
||||
|
||||
e.processor.scheduleServiceElectionRestart(e.retiredCtx, delay, wg, func() {
|
||||
e.startCampaign(wg)
|
||||
})
|
||||
}
|
||||
|
||||
// restartDelayLocked doubles the restart delay for each consecutive
|
||||
// activation failure, capped at serviceElectionRestartMaxDelay, so a
|
||||
// persistently broken Service does not spin the Lease and VIP in a tight
|
||||
// add/delete loop. The caller must hold e.mutex.
|
||||
func (e *serviceElection) restartDelayLocked() time.Duration {
|
||||
delay := serviceElectionRestartBaseDelay
|
||||
for i := 0; i < e.restartFailures && delay < serviceElectionRestartMaxDelay; i++ {
|
||||
delay *= 2
|
||||
}
|
||||
if delay > serviceElectionRestartMaxDelay {
|
||||
delay = serviceElectionRestartMaxDelay
|
||||
}
|
||||
return delay
|
||||
}
|
||||
|
||||
func (p *Processor) scheduleServiceElectionRestart(ctx context.Context, delay time.Duration, wg *sync.WaitGroup, restart func()) {
|
||||
if p.scheduleElectionRestart != nil {
|
||||
p.scheduleElectionRestart(restart)
|
||||
return
|
||||
}
|
||||
wg.Go(func() {
|
||||
timer := time.NewTimer(delay)
|
||||
defer timer.Stop()
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-timer.C:
|
||||
restart()
|
||||
}
|
||||
})
|
||||
}
|
||||
596
pkg/services/election_test.go
Normal file
596
pkg/services/election_test.go
Normal file
@@ -0,0 +1,596 @@
|
||||
package services
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"slices"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kube-vip/kube-vip/pkg/instance"
|
||||
"github.com/kube-vip/kube-vip/pkg/kubevip"
|
||||
"github.com/kube-vip/kube-vip/pkg/lease"
|
||||
"github.com/kube-vip/kube-vip/pkg/servicecontext"
|
||||
v1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"k8s.io/apimachinery/pkg/types"
|
||||
)
|
||||
|
||||
func resetServiceReadiness(t *testing.T, svcCtx *servicecontext.Context) {
|
||||
t.Helper()
|
||||
generation, _, _, ready := svcCtx.ReadinessState()
|
||||
if !ready || !svcCtx.ResetReadinessGeneration(generation) {
|
||||
t.Fatal("Service readiness generation was not reset")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOrderedServiceVIPsUsesCreationTimeAndStableIdentity(t *testing.T) {
|
||||
older := metav1.NewTime(time.Unix(100, 0))
|
||||
newer := metav1.NewTime(time.Unix(200, 0))
|
||||
services := []*v1.Service{
|
||||
{ObjectMeta: metav1.ObjectMeta{Name: "new", Namespace: "default", UID: "new", CreationTimestamp: newer}, Spec: v1.ServiceSpec{LoadBalancerIP: "192.0.2.30"}},
|
||||
{ObjectMeta: metav1.ObjectMeta{Name: "second", Namespace: "default", UID: "second", CreationTimestamp: older}, Spec: v1.ServiceSpec{LoadBalancerIP: "192.0.2.20"}},
|
||||
{ObjectMeta: metav1.ObjectMeta{Name: "first", Namespace: "default", UID: "first", CreationTimestamp: older}, Spec: v1.ServiceSpec{LoadBalancerIP: "192.0.2.10"}},
|
||||
}
|
||||
|
||||
got := orderedServiceVIPs(services)
|
||||
want := []string{"192.0.2.10", "192.0.2.20", "192.0.2.30"}
|
||||
if !slices.Equal(got, want) {
|
||||
t.Fatalf("orderedServiceVIPs() = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestServiceElectionClaimsOnlyCurrentReadinessGeneration(t *testing.T) {
|
||||
p := &Processor{
|
||||
config: &kubevip.Config{},
|
||||
leaseMgr: lease.NewManager(),
|
||||
}
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "service", Namespace: "default", UID: types.UID("service"),
|
||||
}}
|
||||
namespace, name := lease.ServiceName(service)
|
||||
id := lease.NewID(p.config.LeaderElectionType, namespace, name)
|
||||
svcCtx := servicecontext.New(context.Background())
|
||||
defer svcCtx.Cancel()
|
||||
p.svcMap.Store(service.UID, svcCtx)
|
||||
|
||||
svcCtx.SignalReadiness()
|
||||
firstGeneration, _, _, _ := svcCtx.ReadinessState()
|
||||
first, joined := p.joinServiceElection(svcCtx, service, firstGeneration)
|
||||
if !joined {
|
||||
t.Fatal("first ready generation did not join its service election")
|
||||
}
|
||||
if first.claimToken == "" || first.claimToken == string(service.UID) {
|
||||
t.Fatalf("claim token %q is not a coordinator-issued opaque token", first.claimToken)
|
||||
}
|
||||
|
||||
p.leaveServiceElection(first)
|
||||
resetServiceReadiness(t, svcCtx)
|
||||
svcCtx.SignalReadiness()
|
||||
secondGeneration, _, _, _ := svcCtx.ReadinessState()
|
||||
second, joined := p.joinServiceElection(svcCtx, service, secondGeneration)
|
||||
if !joined {
|
||||
t.Fatal("second ready generation did not join its service election")
|
||||
}
|
||||
if second.claimToken == first.claimToken {
|
||||
t.Fatal("new readiness generation reused the old lease claim token")
|
||||
}
|
||||
p.removeServiceElection(first.election)
|
||||
p.electionsMutex.Lock()
|
||||
current := p.elections[id.NamespacedName()]
|
||||
p.electionsMutex.Unlock()
|
||||
if current != second.election {
|
||||
t.Fatal("retired coordinator cleanup removed its replacement")
|
||||
}
|
||||
|
||||
// Delayed cleanup from the old generation must not remove the new claim.
|
||||
p.leaveServiceElection(first)
|
||||
if p.leaseMgr.Get(id) == nil {
|
||||
t.Fatal("stale generation cleanup retired the current service election")
|
||||
}
|
||||
p.leaveServiceElection(second)
|
||||
if p.leaseMgr.Get(id) != nil {
|
||||
t.Fatal("final member withdrawal did not retire its lease")
|
||||
}
|
||||
}
|
||||
|
||||
func TestServiceElectionStaleUIDCannotReleaseRecreatedService(t *testing.T) {
|
||||
p := &Processor{
|
||||
config: &kubevip.Config{},
|
||||
leaseMgr: lease.NewManager(),
|
||||
}
|
||||
annotations := map[string]string{kubevip.ServiceLease: "shared"}
|
||||
oldService := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "service", Namespace: "default", UID: types.UID("old"), Annotations: annotations,
|
||||
}}
|
||||
newService := oldService.DeepCopy()
|
||||
newService.UID = types.UID("new")
|
||||
siblingService := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "sibling", Namespace: "default", UID: types.UID("sibling"), Annotations: annotations,
|
||||
}}
|
||||
|
||||
oldCtx := servicecontext.New(context.Background())
|
||||
siblingCtx := servicecontext.New(context.Background())
|
||||
p.svcMap.Store(oldService.UID, oldCtx)
|
||||
p.svcMap.Store(siblingService.UID, siblingCtx)
|
||||
oldCtx.SignalReadiness()
|
||||
siblingCtx.SignalReadiness()
|
||||
oldGeneration, _, _, _ := oldCtx.ReadinessState()
|
||||
oldMember, joined := p.joinServiceElection(oldCtx, oldService, oldGeneration)
|
||||
if !joined {
|
||||
t.Fatal("old Service did not join its election")
|
||||
}
|
||||
siblingGeneration, _, _, _ := siblingCtx.ReadinessState()
|
||||
siblingMember, joined := p.joinServiceElection(siblingCtx, siblingService, siblingGeneration)
|
||||
if !joined {
|
||||
t.Fatal("sibling Service did not join its election")
|
||||
}
|
||||
originalElection := oldMember.election
|
||||
originalLease := originalElection.lease
|
||||
|
||||
newCtx := servicecontext.New(context.Background())
|
||||
p.svcMap.Store(newService.UID, newCtx)
|
||||
newCtx.SignalReadiness()
|
||||
newGeneration, _, _, _ := newCtx.ReadinessState()
|
||||
newMember, joined := p.joinServiceElection(newCtx, newService, newGeneration)
|
||||
if !joined {
|
||||
t.Fatal("recreated Service did not join its election")
|
||||
}
|
||||
p.leaveServiceElection(oldMember)
|
||||
if newMember.election != originalElection || newMember.election.lease != originalLease {
|
||||
t.Fatal("recreated Service did not join the sibling's live election")
|
||||
}
|
||||
originalElection.mutex.Lock()
|
||||
currentNew := originalElection.members[newService.UID]
|
||||
currentSibling := originalElection.members[siblingService.UID]
|
||||
originalElection.mutex.Unlock()
|
||||
if currentNew != newMember || currentSibling != siblingMember {
|
||||
t.Fatal("stale Service UID cleanup removed a live shared-lease member")
|
||||
}
|
||||
p.leaveServiceElection(newMember)
|
||||
p.leaveServiceElection(siblingMember)
|
||||
}
|
||||
|
||||
func TestServiceElectionActivatesMemberJoiningActiveCampaign(t *testing.T) {
|
||||
p := &Processor{
|
||||
config: &kubevip.Config{EnableServicesElection: true},
|
||||
leaseMgr: lease.NewManager(),
|
||||
}
|
||||
annotations := map[string]string{kubevip.ServiceLease: "shared"}
|
||||
firstService := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "first", Namespace: "default", UID: types.UID("first"), Annotations: annotations,
|
||||
}}
|
||||
secondService := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "second", Namespace: "default", UID: types.UID("second"), Annotations: annotations,
|
||||
}}
|
||||
firstCtx := servicecontext.New(context.Background())
|
||||
secondCtx := servicecontext.New(context.Background())
|
||||
p.svcMap.Store(firstService.UID, firstCtx)
|
||||
p.svcMap.Store(secondService.UID, secondCtx)
|
||||
firstCtx.SignalReadiness()
|
||||
firstGeneration, _, _, _ := firstCtx.ReadinessState()
|
||||
first, joined := p.joinServiceElection(firstCtx, firstService, firstGeneration)
|
||||
if !joined {
|
||||
t.Fatal("first member did not join its election")
|
||||
}
|
||||
|
||||
election := first.election
|
||||
election.mutex.Lock()
|
||||
election.campaign = &serviceElectionCampaign{done: make(chan struct{}), leaderCtx: context.Background()}
|
||||
svcLease := election.lease
|
||||
election.mutex.Unlock()
|
||||
svcLease.ElectionStarted()
|
||||
|
||||
secondCtx.SignalReadiness()
|
||||
secondGeneration, _, _, _ := secondCtx.ReadinessState()
|
||||
second, joined := p.joinServiceElection(secondCtx, secondService, secondGeneration)
|
||||
if !joined {
|
||||
t.Fatal("late member did not join its election")
|
||||
}
|
||||
election.startCampaign(&sync.WaitGroup{})
|
||||
|
||||
election.mutex.Lock()
|
||||
active := second.active
|
||||
election.mutex.Unlock()
|
||||
if !active {
|
||||
t.Fatal("member joining an active campaign was not activated")
|
||||
}
|
||||
election.mutex.Lock()
|
||||
election.campaign = nil
|
||||
election.mutex.Unlock()
|
||||
p.leaveServiceElection(first)
|
||||
p.leaveServiceElection(second)
|
||||
}
|
||||
|
||||
func TestServiceElectionLeadershipLossWaitsForMemberActivation(t *testing.T) {
|
||||
syncStarted := make(chan struct{})
|
||||
releaseSync := make(chan struct{})
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "service", Namespace: "default", UID: types.UID("service"),
|
||||
}}
|
||||
var p *Processor
|
||||
p = &Processor{
|
||||
config: &kubevip.Config{},
|
||||
leaseMgr: lease.NewManager(),
|
||||
serviceSync: func(_ context.Context, _ *servicecontext.Context, _ *v1.Service, _ *sync.WaitGroup, _ bool) error {
|
||||
close(syncStarted)
|
||||
<-releaseSync
|
||||
unlockService := p.lockService(service.UID)
|
||||
p.appendServiceInstance(&instance.Instance{ServiceUID: service.UID, ServiceSnapshot: service.DeepCopy(), AddCalled: true})
|
||||
unlockService()
|
||||
return nil
|
||||
},
|
||||
}
|
||||
svcCtx := servicecontext.New(context.Background())
|
||||
p.svcMap.Store(service.UID, svcCtx)
|
||||
svcCtx.SignalReadiness()
|
||||
generation, _, _, _ := svcCtx.ReadinessState()
|
||||
member, joined := p.joinServiceElection(svcCtx, service, generation)
|
||||
if !joined {
|
||||
t.Fatal("Service did not join its election")
|
||||
}
|
||||
|
||||
election := member.election
|
||||
campaign := &serviceElectionCampaign{done: make(chan struct{})}
|
||||
election.mutex.Lock()
|
||||
election.campaign = campaign
|
||||
svcLease := election.lease
|
||||
election.mutex.Unlock()
|
||||
svcLease.ElectionStarted()
|
||||
|
||||
activationDone := make(chan struct{})
|
||||
go func() {
|
||||
election.activateMember(context.Background(), member, svcLease, campaign, &sync.WaitGroup{})
|
||||
close(activationDone)
|
||||
}()
|
||||
<-syncStarted
|
||||
|
||||
stopDone := make(chan struct{})
|
||||
go func() {
|
||||
election.stopCampaign(svcLease, campaign)
|
||||
close(stopDone)
|
||||
}()
|
||||
waitForCondition(t, func() bool {
|
||||
election.mutex.Lock()
|
||||
defer election.mutex.Unlock()
|
||||
return campaign.stopped
|
||||
}, "campaign leadership loss")
|
||||
select {
|
||||
case <-stopDone:
|
||||
t.Fatal("campaign stop completed before in-flight member activation")
|
||||
default:
|
||||
}
|
||||
|
||||
close(releaseSync)
|
||||
select {
|
||||
case <-activationDone:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("member activation did not finish")
|
||||
}
|
||||
select {
|
||||
case <-stopDone:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("campaign stop did not finish after member activation")
|
||||
}
|
||||
|
||||
if svcLease.Elected.Load() {
|
||||
t.Fatal("lease remained elected after campaign stop")
|
||||
}
|
||||
election.mutex.Lock()
|
||||
active := member.active
|
||||
election.mutex.Unlock()
|
||||
if active {
|
||||
t.Fatal("member remained active after campaign stop")
|
||||
}
|
||||
if p.findServiceInstance(service) != nil {
|
||||
t.Fatal("in-flight activation left a tracked Service instance after leadership loss")
|
||||
}
|
||||
p.leaveServiceElection(member)
|
||||
}
|
||||
|
||||
func TestServiceElectionLeadershipLossCancelsMemberActivation(t *testing.T) {
|
||||
syncStarted := make(chan struct{})
|
||||
cancellationObserved := make(chan struct{})
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "service", Namespace: "default", UID: types.UID("service"),
|
||||
}}
|
||||
p := &Processor{
|
||||
config: &kubevip.Config{},
|
||||
leaseMgr: lease.NewManager(),
|
||||
serviceSync: func(ctx context.Context, _ *servicecontext.Context, _ *v1.Service, _ *sync.WaitGroup, _ bool) error {
|
||||
close(syncStarted)
|
||||
<-ctx.Done()
|
||||
close(cancellationObserved)
|
||||
return ctx.Err()
|
||||
},
|
||||
}
|
||||
svcCtx := servicecontext.New(context.Background())
|
||||
p.svcMap.Store(service.UID, svcCtx)
|
||||
svcCtx.SignalReadiness()
|
||||
generation, _, _, _ := svcCtx.ReadinessState()
|
||||
member, joined := p.joinServiceElection(svcCtx, service, generation)
|
||||
if !joined {
|
||||
t.Fatal("Service did not join its election")
|
||||
}
|
||||
|
||||
election := member.election
|
||||
leaderCtx, cancelLeader := context.WithCancel(context.Background())
|
||||
campaign := &serviceElectionCampaign{done: make(chan struct{}), leaderCtx: leaderCtx, cancelLeader: cancelLeader}
|
||||
election.mutex.Lock()
|
||||
election.campaign = campaign
|
||||
svcLease := election.lease
|
||||
election.mutex.Unlock()
|
||||
svcLease.ElectionStarted()
|
||||
|
||||
activationDone := make(chan struct{})
|
||||
go func() {
|
||||
election.activateMember(leaderCtx, member, svcLease, campaign, &sync.WaitGroup{})
|
||||
close(activationDone)
|
||||
}()
|
||||
<-syncStarted
|
||||
|
||||
stopDone := make(chan struct{})
|
||||
go func() {
|
||||
election.stopCampaign(svcLease, campaign)
|
||||
close(stopDone)
|
||||
}()
|
||||
select {
|
||||
case <-cancellationObserved:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("leadership loss did not cancel member activation")
|
||||
}
|
||||
select {
|
||||
case <-activationDone:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("member activation did not finish after cancellation")
|
||||
}
|
||||
select {
|
||||
case <-stopDone:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("campaign stop did not finish after activation cancellation")
|
||||
}
|
||||
p.leaveServiceElection(member)
|
||||
}
|
||||
|
||||
func TestServiceElectionActivationFailureAfterMemberRemovalDoesNotPanic(t *testing.T) {
|
||||
syncStarted := make(chan struct{})
|
||||
releaseSync := make(chan struct{})
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "service", Namespace: "default", UID: types.UID("service"),
|
||||
}}
|
||||
p := &Processor{
|
||||
config: &kubevip.Config{},
|
||||
leaseMgr: lease.NewManager(),
|
||||
serviceSync: func(_ context.Context, _ *servicecontext.Context, _ *v1.Service, _ *sync.WaitGroup, _ bool) error {
|
||||
close(syncStarted)
|
||||
<-releaseSync
|
||||
return errors.New("activation failed")
|
||||
},
|
||||
}
|
||||
svcCtx := servicecontext.New(context.Background())
|
||||
p.svcMap.Store(service.UID, svcCtx)
|
||||
svcCtx.SignalReadiness()
|
||||
generation, _, _, _ := svcCtx.ReadinessState()
|
||||
member, joined := p.joinServiceElection(svcCtx, service, generation)
|
||||
if !joined {
|
||||
t.Fatal("Service did not join its election")
|
||||
}
|
||||
|
||||
election := member.election
|
||||
campaign := &serviceElectionCampaign{done: make(chan struct{})}
|
||||
election.mutex.Lock()
|
||||
election.campaign = campaign
|
||||
svcLease := election.lease
|
||||
election.mutex.Unlock()
|
||||
svcLease.ElectionStarted()
|
||||
|
||||
activationDone := make(chan any, 1)
|
||||
go func() {
|
||||
defer func() {
|
||||
activationDone <- recover()
|
||||
}()
|
||||
election.activateMember(context.Background(), member, svcLease, campaign, &sync.WaitGroup{})
|
||||
}()
|
||||
<-syncStarted
|
||||
|
||||
p.leaveServiceElection(member)
|
||||
close(campaign.done)
|
||||
close(releaseSync)
|
||||
|
||||
select {
|
||||
case panicValue := <-activationDone:
|
||||
if panicValue != nil {
|
||||
t.Fatalf("activation failure after member removal panicked: %v", panicValue)
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("member activation did not finish")
|
||||
}
|
||||
}
|
||||
|
||||
func TestStopCampaignDoesNotCleanupInactiveMember(t *testing.T) {
|
||||
p := &Processor{
|
||||
config: &kubevip.Config{},
|
||||
leaseMgr: lease.NewManager(),
|
||||
}
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "inactive", Namespace: "default", UID: types.UID("inactive"),
|
||||
}}
|
||||
svcCtx := servicecontext.New(context.Background())
|
||||
p.svcMap.Store(service.UID, svcCtx)
|
||||
svcCtx.SignalReadiness()
|
||||
generation, _, _, _ := svcCtx.ReadinessState()
|
||||
member, joined := p.joinServiceElection(svcCtx, service, generation)
|
||||
if !joined {
|
||||
t.Fatal("inactive Service did not join its election")
|
||||
}
|
||||
serviceInstance := &instance.Instance{ServiceUID: service.UID, ServiceSnapshot: service.DeepCopy()}
|
||||
p.ServiceInstances = []*instance.Instance{serviceInstance}
|
||||
|
||||
election := member.election
|
||||
campaign := &serviceElectionCampaign{done: make(chan struct{})}
|
||||
election.mutex.Lock()
|
||||
election.campaign = campaign
|
||||
svcLease := election.lease
|
||||
election.mutex.Unlock()
|
||||
svcLease.ElectionStarted()
|
||||
|
||||
election.stopCampaign(svcLease, campaign)
|
||||
if got := p.findServiceInstance(service); got != serviceInstance {
|
||||
t.Fatal("campaign stop detached an inactive member's instance")
|
||||
}
|
||||
election.mutex.Lock()
|
||||
active := member.active
|
||||
election.mutex.Unlock()
|
||||
if active {
|
||||
t.Fatal("inactive member became active during campaign stop")
|
||||
}
|
||||
p.leaveServiceElection(member)
|
||||
}
|
||||
|
||||
func TestCancelledContextDrainsActiveMemberBeforeReplacement(t *testing.T) {
|
||||
p := &Processor{config: &kubevip.Config{}, leaseMgr: lease.NewManager()}
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "service", Namespace: "default", UID: types.UID("service"),
|
||||
}}
|
||||
svcCtx := servicecontext.New(context.Background())
|
||||
if !svcCtx.StartWatching() {
|
||||
t.Fatal("watcher ownership was not acquired")
|
||||
}
|
||||
p.svcMap.Store(service.UID, svcCtx)
|
||||
p.ServiceInstances = []*instance.Instance{{ServiceUID: service.UID, ServiceSnapshot: service.DeepCopy(), AddCalled: true}}
|
||||
svcCtx.SignalReadiness()
|
||||
generation, _, _, _ := svcCtx.ReadinessState()
|
||||
member, joined := p.joinServiceElection(svcCtx, service, generation)
|
||||
if !joined {
|
||||
t.Fatal("Service did not join election")
|
||||
}
|
||||
member.election.mutex.Lock()
|
||||
member.active = true
|
||||
member.election.mutex.Unlock()
|
||||
svcCtx.Cancel()
|
||||
|
||||
replacement := make(chan *servicecontext.Context, 1)
|
||||
errs := make(chan error, 1)
|
||||
go func() {
|
||||
current, err := p.ensureServiceContext(context.Background(), service)
|
||||
if err != nil {
|
||||
errs <- err
|
||||
return
|
||||
}
|
||||
replacement <- current
|
||||
}()
|
||||
select {
|
||||
case <-replacement:
|
||||
t.Fatal("replacement was created before active-member cleanup")
|
||||
case err := <-errs:
|
||||
t.Fatalf("ensureServiceContext() error = %v", err)
|
||||
case <-time.After(20 * time.Millisecond):
|
||||
}
|
||||
|
||||
member.election.deactivateMember(member)
|
||||
p.leaveServiceElection(member)
|
||||
svcCtx.StopWatching()
|
||||
|
||||
select {
|
||||
case current := <-replacement:
|
||||
if current == svcCtx || current.Ctx.Err() != nil {
|
||||
t.Fatal("replacement context is not live")
|
||||
}
|
||||
case err := <-errs:
|
||||
t.Fatalf("ensureServiceContext() error = %v", err)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("replacement was not created after cleanup drained")
|
||||
}
|
||||
if p.findServiceInstance(service) != nil {
|
||||
t.Fatal("active member datapath was not cleaned before replacement")
|
||||
}
|
||||
}
|
||||
|
||||
func TestServiceElectionDelayedCleanupDoesNotDeleteNewReadinessGeneration(t *testing.T) {
|
||||
p := &Processor{
|
||||
config: &kubevip.Config{},
|
||||
leaseMgr: lease.NewManager(),
|
||||
}
|
||||
service := &v1.Service{ObjectMeta: metav1.ObjectMeta{
|
||||
Name: "service", Namespace: "default", UID: types.UID("service"),
|
||||
}}
|
||||
svcCtx := servicecontext.New(context.Background())
|
||||
p.svcMap.Store(service.UID, svcCtx)
|
||||
p.ServiceInstances = []*instance.Instance{{ServiceUID: service.UID, ServiceSnapshot: service.DeepCopy()}}
|
||||
svcCtx.SignalReadiness()
|
||||
firstGeneration, _, _, _ := svcCtx.ReadinessState()
|
||||
first, joined := p.joinServiceElection(svcCtx, service, firstGeneration)
|
||||
if !joined {
|
||||
t.Fatal("first readiness generation did not join")
|
||||
}
|
||||
|
||||
resetServiceReadiness(t, svcCtx)
|
||||
svcCtx.SignalReadiness()
|
||||
secondGeneration, _, _, _ := svcCtx.ReadinessState()
|
||||
second, joined := p.joinServiceElection(svcCtx, service, secondGeneration)
|
||||
if !joined {
|
||||
t.Fatal("second readiness generation did not join")
|
||||
}
|
||||
if err := p.onStoppedLeadingMember(first, first.election.lease); err != nil {
|
||||
t.Fatalf("delayed old-generation cleanup returned an error: %v", err)
|
||||
}
|
||||
if p.findServiceInstance(service) == nil {
|
||||
t.Fatal("delayed old-generation cleanup deleted the current Service instance")
|
||||
}
|
||||
p.leaveServiceElection(second)
|
||||
}
|
||||
|
||||
// TestServiceElectionRestartDelayBacksOffOnRepeatedFailuresAndResets guards
|
||||
// against a persistent activation failure turning into a tight Lease
|
||||
// acquire/release and VIP add/delete loop: the restart delay must grow after
|
||||
// a failure and drop back to the base delay once activation succeeds.
|
||||
func TestServiceElectionRestartDelayBacksOffOnRepeatedFailuresAndResets(t *testing.T) {
|
||||
id := lease.NewID("kubernetes", "default", "svc")
|
||||
svcLease := lease.NewManager().Add(context.Background(), id)
|
||||
campaign := &serviceElectionCampaign{done: make(chan struct{})}
|
||||
election := &serviceElection{lease: svcLease, campaign: campaign}
|
||||
|
||||
baseDelay := serviceElectionRestartBaseDelay
|
||||
election.cancelCampaign(svcLease, campaign)
|
||||
|
||||
election.mutex.Lock()
|
||||
afterFailure := election.restartDelayLocked()
|
||||
election.mutex.Unlock()
|
||||
if afterFailure <= baseDelay {
|
||||
t.Fatalf("restart delay did not grow after a failure: base=%v after=%v", baseDelay, afterFailure)
|
||||
}
|
||||
|
||||
election.resetRestartFailures()
|
||||
election.mutex.Lock()
|
||||
afterReset := election.restartDelayLocked()
|
||||
election.mutex.Unlock()
|
||||
if afterReset != baseDelay {
|
||||
t.Fatalf("restart delay was not reset after a successful activation: got=%v want=%v", afterReset, baseDelay)
|
||||
}
|
||||
}
|
||||
|
||||
func TestServiceElectionRestartTimerStopsOnRetirement(t *testing.T) {
|
||||
p := &Processor{}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
var wg sync.WaitGroup
|
||||
restarted := make(chan struct{}, 1)
|
||||
p.scheduleServiceElectionRestart(ctx, time.Hour, &wg, func() { restarted <- struct{}{} })
|
||||
cancel()
|
||||
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
wg.Wait()
|
||||
close(done)
|
||||
}()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("restart timer did not drain after coordinator retirement")
|
||||
}
|
||||
select {
|
||||
case <-restarted:
|
||||
t.Fatal("restart ran after coordinator retirement")
|
||||
default:
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user