flock PR validation / validate (pull_request) Successful in 11s
Three defects enabled the 2026-08-16 Gitea blackhole (bug-wdgjpz3a00gd): 1. Orphaned allocation GC missing: ungraceful eviction (TaintManagerEviction) never calls CNI DEL, so the old node keeps advertising the pod's public /128 via BGP. Older allocation wins BGP path selection; live pod's node yields → blackhole. Fix: after the pod informer syncs at startup, sweep all committed allocations via orphanedCommitted(). Any allocation whose owner pod is absent from the node (or whose UID mismatches, indicating name reuse) is torn down, removed from the store, and released from IPAM. A 60 s periodic GC goroutine provides the same sweep while the agent runs. 2. renderBird outside-aggregate IP loop lacked pod liveness check: stale committed allocations caused BIRD to keep advertising the /128 even in steady state between GC ticks. Fix: before adding an outside-aggregate primary IP to the BIRD export, verify the pod is still in the node-scoped informer cache with a matching UID. Orphans are skipped silently; the GC cleans them on the next tick. 3. birdc startup race: the agent's first Render() fires before BIRD has bound /run/flock/bird.ctl, so the configure call silently fails with "Unable to connect" and the initial routes are never advertised. A container-only flock-agent restart (BIRD left running) avoids the race; a full pod restart re-hits it. Fix: reload() now retries up to 20 × 500 ms on socket-absent and "Unable to connect" conditions. Any other birdc failure (syntax error, etc.) is not retried. Fixes bug-wdgjpz3a00gd Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
242 lines
7.7 KiB
Go
242 lines
7.7 KiB
Go
//go:build linux
|
|
|
|
package agent
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"net"
|
|
"os"
|
|
"time"
|
|
|
|
"code.fritzlab.net/fritzlab/flock/pkg/agent/netpol"
|
|
)
|
|
|
|
// hostMultipathHashSysctls is the set of node-level sysctls flock-agent
|
|
// best-effort writes at startup. Default policy 0 hashes only on
|
|
// (saddr, daddr); policy 1 adds L4 (sport, dport, proto), giving real
|
|
// per-connection ECMP across multipath nexthops — required for sensible
|
|
// distribution across multiple anycast pods on the same node.
|
|
var hostMultipathHashSysctls = map[string]string{
|
|
"/proc/sys/net/ipv4/fib_multipath_hash_policy": "1",
|
|
"/proc/sys/net/ipv6/fib_multipath_hash_policy": "1",
|
|
}
|
|
|
|
// applyHostSysctls writes the sysctls in m, logging but not failing on
|
|
// errors. flock-agent is privileged so this works in the production
|
|
// DaemonSet; in environments where it doesn't, single-pod-per-node
|
|
// anycast still works (this only affects the multi-pod-per-node case).
|
|
func applyHostSysctls(s *Server) {
|
|
for path, value := range hostMultipathHashSysctls {
|
|
if err := os.WriteFile(path, []byte(value), 0o644); err != nil {
|
|
s.Logger.Warn("set host sysctl", "path", path, "value", value, "err", err)
|
|
continue
|
|
}
|
|
s.Logger.Info("host sysctl set", "path", path, "value", value)
|
|
}
|
|
}
|
|
|
|
// configureRuntime wires Pod informer, IPAM, netlink, and BIRD on a real
|
|
// Linux node. Steps:
|
|
//
|
|
// 1. Wait for NodeConfig (operator-applied per-node CR).
|
|
// 2. Reconcile any pre-existing kernel state from allocations.json into
|
|
// IPAM.used (so we never re-allocate an in-flight pod's IP).
|
|
// 3. Garbage-collect any state==pending entries (partial ADDs from a
|
|
// previous agent generation).
|
|
// 4. Start the Pod informer (filtered to spec.nodeName == node).
|
|
// 5. Build PodHandler and SetHandlers(add, del, check).
|
|
// 6. Install BIRD blackhole summary routes + render initial config.
|
|
func (s *Server) configureRuntime(ctx context.Context) error {
|
|
applyHostSysctls(s)
|
|
|
|
if err := s.firstAvailableNodeConfig(ctx, 60*time.Second); err != nil {
|
|
return err
|
|
}
|
|
nc := s.NodeConfig.Load()
|
|
|
|
ipam, err := NewIPAM(nc.Spec.CIDR6, nc.Spec.CIDR4)
|
|
if err != nil {
|
|
return fmt.Errorf("init ipam: %w", err)
|
|
}
|
|
|
|
// Reconcile committed entries; GC pending entries.
|
|
for _, a := range s.Store.Snapshot() {
|
|
switch a.State {
|
|
case StateCommitted:
|
|
if a.IP6 != "" {
|
|
ipam.MarkInUse(net.ParseIP(a.IP6))
|
|
}
|
|
if a.IP4 != "" {
|
|
ipam.MarkInUse(net.ParseIP(a.IP4))
|
|
}
|
|
case StatePending:
|
|
s.Logger.Info("GC pending allocation", "container_id", a.ContainerID)
|
|
_ = Teardown(a.ContainerID, net.ParseIP(a.IP6), net.ParseIP(a.IP4))
|
|
_ = s.Store.Delete(a.ContainerID)
|
|
}
|
|
}
|
|
|
|
pods, err := StartPodInformer(ctx, s.restCfg, s.Node, s.Logger)
|
|
if err != nil {
|
|
return fmt.Errorf("pod informer: %w", err)
|
|
}
|
|
|
|
// Startup orphan GC: the pod informer is now fully synced. Walk all
|
|
// committed allocations and release any whose owner pod is absent from
|
|
// this node. This catches ungraceful evictions where CNI DEL never ran
|
|
// (TaintManagerEviction path) and prevents stale public /128s from
|
|
// suppressing the live pod's BGP advertisement after rescheduling.
|
|
gcOrphans := func(label string) int {
|
|
orphans := orphanedCommitted(s.Store.Snapshot(), func(ns, name string) (string, bool) {
|
|
pod, ok := pods.Get(ns, name)
|
|
if !ok {
|
|
return "", false
|
|
}
|
|
return string(pod.UID), true
|
|
})
|
|
for _, a := range orphans {
|
|
s.Logger.Info(label,
|
|
"container_id", a.ContainerID,
|
|
"pod", a.Namespace+"/"+a.PodName,
|
|
"ip6", a.IP6,
|
|
"ip4", a.IP4,
|
|
)
|
|
_ = Teardown(a.ContainerID, net.ParseIP(a.IP6), net.ParseIP(a.IP4))
|
|
_ = s.Store.Delete(a.ContainerID)
|
|
ipam.Release(net.ParseIP(a.IP6), net.ParseIP(a.IP4))
|
|
}
|
|
return len(orphans)
|
|
}
|
|
gcOrphans("GC orphaned committed allocation (startup)")
|
|
|
|
// Keep NetworkUnavailable=False so the node.kubernetes.io/network-
|
|
// unavailable taint never gets re-applied. Calico's calico-node sets
|
|
// it on shutdown; without an owner replacing it, kubelet's controller
|
|
// taints the node and blocks scheduling.
|
|
go keepNetworkAvailable(ctx, s.restCfg, s.Node, s.Logger)
|
|
|
|
bird := &BirdManager{
|
|
NodeName: s.Node,
|
|
ConfigPath: "/etc/flock/bird/bird.conf",
|
|
BirdcSocket: "/run/flock/bird.ctl",
|
|
Logger: s.Logger,
|
|
}
|
|
// Install kernel blackhole routes for the node summary CIDRs. These
|
|
// stay regardless of BGP — they keep the kernel from sending unknown
|
|
// destinations within our /64 to a default route loop.
|
|
if err := bird.SummaryRoutes(nc); err != nil {
|
|
s.Logger.Warn("install summary routes", "err", err)
|
|
}
|
|
// Calico is fenced off this node (Tigera Installation CR adds a
|
|
// nodeAffinity excluding flock.fritzlab.net/agent on
|
|
// calicoNodeDaemonSet). flock now owns BGP from this host.
|
|
routerID := routerIDFromNodeIP(s.restCfg)
|
|
if err := bird.Render(nc, nil, nil, routerID); err != nil {
|
|
s.Logger.Warn("initial bird render", "err", err)
|
|
}
|
|
|
|
// AnycastReconciler is the single owner of bird re-renders going
|
|
// forward. It runs every 2s + on Pod readiness changes + on each
|
|
// successful CNI ADD/DEL.
|
|
anycast := NewAnycastReconciler(s.Node, s.Store, pods, s.NodeConfig, bird, routerID, s.Logger)
|
|
pods.OnReadyChange(anycast.Trigger)
|
|
go anycast.Run(ctx)
|
|
|
|
// Background tick for SummaryRoutes (idempotent) in case the kernel
|
|
// blackhole disappears for any reason.
|
|
go func() {
|
|
t := time.NewTicker(60 * time.Second)
|
|
defer t.Stop()
|
|
for {
|
|
select {
|
|
case <-ctx.Done():
|
|
return
|
|
case <-t.C:
|
|
if cur := s.NodeConfig.Load(); cur != nil {
|
|
_ = bird.SummaryRoutes(cur)
|
|
}
|
|
}
|
|
}
|
|
}()
|
|
|
|
// Periodic orphan GC: defense-in-depth against allocations that escape
|
|
// the startup sweep (e.g. a pod evicted while the agent is running and
|
|
// the CNI DEL is never delivered). Keeps the store and IPAM in sync
|
|
// with the live pod set without requiring a full agent restart.
|
|
go func() {
|
|
t := time.NewTicker(60 * time.Second)
|
|
defer t.Stop()
|
|
for {
|
|
select {
|
|
case <-ctx.Done():
|
|
return
|
|
case <-t.C:
|
|
if n := gcOrphans("GC orphaned committed allocation (periodic)"); n > 0 {
|
|
anycast.Trigger()
|
|
}
|
|
}
|
|
}
|
|
}()
|
|
|
|
// NetworkPolicy enforcement.
|
|
world := netpol.NewWorld(s.Logger)
|
|
if err := world.Start(ctx, s.restCfg); err != nil {
|
|
return fmt.Errorf("netpol informers: %w", err)
|
|
}
|
|
npApplier := &netpol.Applier{}
|
|
npReconciler := netpol.NewReconciler(world, func() []netpol.Pod {
|
|
return collectLocalPods(s.Store, pods)
|
|
}, npApplier, s.Logger)
|
|
go npReconciler.Run(ctx)
|
|
|
|
handler := &PodHandler{
|
|
Node: s.Node,
|
|
Store: s.Store,
|
|
IPAM: ipam,
|
|
Pods: pods,
|
|
NodeConfig: s.NodeConfig,
|
|
Logger: s.Logger,
|
|
SetupFunc: Setup,
|
|
TeardownFunc: Teardown,
|
|
AfterCommit: func() {
|
|
anycast.Trigger()
|
|
// Re-evaluate policy on every CNI ADD/DEL so a brand-new
|
|
// pod's chain lands before its first packet egresses.
|
|
npReconciler.Trigger()
|
|
},
|
|
}
|
|
s.RPC.SetHandlers(handler.Add, handler.Del, handler.Check)
|
|
s.Logger.Info("runtime ready",
|
|
"asn", nc.Spec.BGP.ASN,
|
|
"cidr6", nc.Spec.CIDR6,
|
|
"cidr4", nc.Spec.CIDR4,
|
|
"committed", len(s.Store.Snapshot()),
|
|
)
|
|
return nil
|
|
}
|
|
|
|
// routerIDFromNodeIP picks a stable IPv4 to use as BIRD router-id. Uses
|
|
// the host network for now; falls back to a synthesized value derived
|
|
// from the node name if no v4 is reachable.
|
|
func routerIDFromNodeIP(_ interface{}) string {
|
|
// Best-effort: read the kernel route table for a default-route src.
|
|
addrs, err := net.InterfaceAddrs()
|
|
if err == nil {
|
|
for _, a := range addrs {
|
|
ipn, ok := a.(*net.IPNet)
|
|
if !ok {
|
|
continue
|
|
}
|
|
v4 := ipn.IP.To4()
|
|
if v4 == nil || v4.IsLoopback() || v4.IsLinkLocalUnicast() {
|
|
continue
|
|
}
|
|
return v4.String()
|
|
}
|
|
}
|
|
// Fallback: 127.0.0.1 — bird will accept it but BGP peers won't like a
|
|
// duplicate router-id. The agent log will scream above this if it fires.
|
|
return "127.0.0.1"
|
|
}
|