mirror of
https://git.vectorsigma.ru/public/k3s.git
synced 2026-08-04 10:09:45 +00:00
Refactor egress-selector pods mode to watch pods
Watching pods appears to be the most reliable way to ensure that the proxy routes and authorizes connections. Signed-off-by: Brad Davidson <brad.davidson@rancher.com>
This commit is contained in:
committed by
Brad Davidson
parent
8456d98283
commit
15b8fb962a
@@ -15,7 +15,6 @@ import (
|
||||
daemonconfig "github.com/k3s-io/k3s/pkg/daemons/config"
|
||||
"github.com/k3s-io/k3s/pkg/util"
|
||||
"github.com/k3s-io/k3s/pkg/version"
|
||||
"github.com/pkg/errors"
|
||||
"github.com/rancher/remotedialer"
|
||||
"github.com/sirupsen/logrus"
|
||||
"github.com/yl2chen/cidranger"
|
||||
@@ -29,14 +28,27 @@ import (
|
||||
"k8s.io/client-go/tools/cache"
|
||||
"k8s.io/client-go/tools/clientcmd"
|
||||
toolswatch "k8s.io/client-go/tools/watch"
|
||||
"k8s.io/kubectl/pkg/util/podutils"
|
||||
)
|
||||
|
||||
var (
|
||||
ports = map[string]bool{
|
||||
"10250": true,
|
||||
"10010": true,
|
||||
}
|
||||
)
|
||||
type agentTunnel struct {
|
||||
client kubernetes.Interface
|
||||
cidrs cidranger.Ranger
|
||||
ports map[string]bool
|
||||
mode string
|
||||
}
|
||||
|
||||
// explicit interface check
|
||||
var _ cidranger.RangerEntry = &podEntry{}
|
||||
|
||||
type podEntry struct {
|
||||
cidr net.IPNet
|
||||
hostNet bool
|
||||
}
|
||||
|
||||
func (p *podEntry) Network() net.IPNet {
|
||||
return p.cidr
|
||||
}
|
||||
|
||||
func Setup(ctx context.Context, config *daemonconfig.Node, proxy proxy.Proxy) error {
|
||||
restConfig, err := clientcmd.BuildConfigFromFlags("", config.AgentConfig.KubeConfigK3sController)
|
||||
@@ -62,14 +74,25 @@ func Setup(ctx context.Context, config *daemonconfig.Node, proxy proxy.Proxy) er
|
||||
tunnel := &agentTunnel{
|
||||
client: client,
|
||||
cidrs: cidranger.NewPCTrieRanger(),
|
||||
ports: map[string]bool{},
|
||||
mode: config.EgressSelectorMode,
|
||||
}
|
||||
|
||||
if tunnel.mode == daemonconfig.EgressSelectorModeCluster {
|
||||
for _, cidr := range config.AgentConfig.ClusterCIDRs {
|
||||
logrus.Infof("Tunnel authorizer added Cluster CIDR %s", cidr)
|
||||
tunnel.cidrs.Insert(cidranger.NewBasicRangerEntry(*cidr))
|
||||
apiServerReady := make(chan struct{})
|
||||
go func() {
|
||||
if err := util.WaitForAPIServerReady(ctx, config.AgentConfig.KubeConfigKubelet, util.DefaultAPIServerReadyTimeout); err != nil {
|
||||
logrus.Fatalf("Tunnel watches failed to wait for apiserver ready: %v", err)
|
||||
}
|
||||
close(apiServerReady)
|
||||
}()
|
||||
|
||||
switch tunnel.mode {
|
||||
case daemonconfig.EgressSelectorModeCluster:
|
||||
// In Cluster mode, we allow the cluster CIDRs, and any connections to the node's IPs for pods using host network.
|
||||
tunnel.clusterAuth(config)
|
||||
case daemonconfig.EgressSelectorModePod:
|
||||
// In Pod mode, we watch pods assigned to this node, and allow their addresses, as well as ports used by containers with host network.
|
||||
go tunnel.watchPods(ctx, apiServerReady, config)
|
||||
}
|
||||
|
||||
// The loadbalancer is only disabled when there is a local apiserver. Servers without a local
|
||||
@@ -93,79 +116,9 @@ func Setup(ctx context.Context, config *daemonconfig.Node, proxy proxy.Proxy) er
|
||||
}
|
||||
}
|
||||
|
||||
// Attempt to connect to supervisors, storing their cancellation function for later when we
|
||||
// need to disconnect.
|
||||
disconnect := map[string]context.CancelFunc{}
|
||||
wg := &sync.WaitGroup{}
|
||||
for _, address := range proxy.SupervisorAddresses() {
|
||||
if _, ok := disconnect[address]; !ok {
|
||||
disconnect[address] = tunnel.connect(ctx, wg, address, tlsConfig)
|
||||
}
|
||||
}
|
||||
|
||||
// Once the apiserver is up, go into a watch loop, adding and removing tunnels as endpoints come
|
||||
// and go from the cluster.
|
||||
go func() {
|
||||
if err := util.WaitForAPIServerReady(ctx, config.AgentConfig.KubeConfigKubelet, util.DefaultAPIServerReadyTimeout); err != nil {
|
||||
logrus.Warnf("Tunnel endpoint watch failed to wait for apiserver ready: %v", err)
|
||||
}
|
||||
|
||||
endpoints := client.CoreV1().Endpoints(metav1.NamespaceDefault)
|
||||
fieldSelector := fields.Set{metav1.ObjectNameField: "kubernetes"}.String()
|
||||
lw := &cache.ListWatch{
|
||||
ListFunc: func(options metav1.ListOptions) (object runtime.Object, e error) {
|
||||
options.FieldSelector = fieldSelector
|
||||
return endpoints.List(ctx, options)
|
||||
},
|
||||
WatchFunc: func(options metav1.ListOptions) (i watch.Interface, e error) {
|
||||
options.FieldSelector = fieldSelector
|
||||
return endpoints.Watch(ctx, options)
|
||||
},
|
||||
}
|
||||
|
||||
_, _, watch, done := toolswatch.NewIndexerInformerWatcher(lw, &v1.Endpoints{})
|
||||
|
||||
defer func() {
|
||||
watch.Stop()
|
||||
<-done
|
||||
}()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case ev, ok := <-watch.ResultChan():
|
||||
endpoint, ok := ev.Object.(*v1.Endpoints)
|
||||
if !ok {
|
||||
logrus.Errorf("Tunnel watch failed: event object not of type v1.Endpoints")
|
||||
continue
|
||||
}
|
||||
|
||||
newAddresses := util.GetAddresses(endpoint)
|
||||
if reflect.DeepEqual(newAddresses, proxy.SupervisorAddresses()) {
|
||||
continue
|
||||
}
|
||||
proxy.Update(newAddresses)
|
||||
|
||||
validEndpoint := map[string]bool{}
|
||||
|
||||
for _, address := range proxy.SupervisorAddresses() {
|
||||
validEndpoint[address] = true
|
||||
if _, ok := disconnect[address]; !ok {
|
||||
disconnect[address] = tunnel.connect(ctx, nil, address, tlsConfig)
|
||||
}
|
||||
}
|
||||
|
||||
for address, cancel := range disconnect {
|
||||
if !validEndpoint[address] {
|
||||
cancel()
|
||||
delete(disconnect, address)
|
||||
logrus.Infof("Stopped tunnel to %s", address)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}()
|
||||
go tunnel.watchEndpoints(ctx, apiServerReady, wg, tlsConfig, proxy)
|
||||
|
||||
wait := make(chan int, 1)
|
||||
go func() {
|
||||
@@ -183,11 +136,166 @@ func Setup(ctx context.Context, config *daemonconfig.Node, proxy proxy.Proxy) er
|
||||
return nil
|
||||
}
|
||||
|
||||
type agentTunnel struct {
|
||||
sync.Mutex
|
||||
client kubernetes.Interface
|
||||
cidrs cidranger.Ranger
|
||||
mode string
|
||||
func (a *agentTunnel) clusterAuth(config *daemonconfig.Node) {
|
||||
// In Cluster mode, we add static entries for the Node IPs and Cluster CIDRs
|
||||
for _, ip := range config.AgentConfig.NodeIPs {
|
||||
if cidr, err := util.IPToIPNet(ip); err == nil {
|
||||
logrus.Infof("Tunnel authorizer adding Node IP %s", cidr)
|
||||
a.cidrs.Insert(&podEntry{cidr: *cidr})
|
||||
}
|
||||
}
|
||||
for _, cidr := range config.AgentConfig.ClusterCIDRs {
|
||||
logrus.Infof("Tunnel authorizer adding Cluster CIDR %s", cidr)
|
||||
a.cidrs.Insert(&podEntry{cidr: *cidr})
|
||||
}
|
||||
}
|
||||
|
||||
// watchPods watches for pods assigned to this node, adding their IPs to the CIDR list.
|
||||
// If the pod uses host network, we instead add the
|
||||
func (a *agentTunnel) watchPods(ctx context.Context, apiServerReady <-chan struct{}, config *daemonconfig.Node) {
|
||||
for _, ip := range config.AgentConfig.NodeIPs {
|
||||
if cidr, err := util.IPToIPNet(ip); err == nil {
|
||||
logrus.Infof("Tunnel authorizer adding Node IP %s", cidr)
|
||||
a.cidrs.Insert(&podEntry{cidr: *cidr, hostNet: true})
|
||||
}
|
||||
}
|
||||
|
||||
<-apiServerReady
|
||||
|
||||
nodeName := os.Getenv("NODE_NAME")
|
||||
pods := a.client.CoreV1().Pods(metav1.NamespaceNone)
|
||||
fieldSelector := fields.Set{"spec.nodeName": nodeName}.String()
|
||||
lw := &cache.ListWatch{
|
||||
ListFunc: func(options metav1.ListOptions) (object runtime.Object, e error) {
|
||||
options.FieldSelector = fieldSelector
|
||||
return pods.List(ctx, options)
|
||||
},
|
||||
WatchFunc: func(options metav1.ListOptions) (i watch.Interface, e error) {
|
||||
options.FieldSelector = fieldSelector
|
||||
return pods.Watch(ctx, options)
|
||||
},
|
||||
}
|
||||
|
||||
logrus.Infof("Tunnnel authorizer watching Pods")
|
||||
_, _, watch, done := toolswatch.NewIndexerInformerWatcher(lw, &v1.Pod{})
|
||||
|
||||
defer func() {
|
||||
watch.Stop()
|
||||
<-done
|
||||
}()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case ev, ok := <-watch.ResultChan():
|
||||
pod, ok := ev.Object.(*v1.Pod)
|
||||
if !ok {
|
||||
logrus.Errorf("Tunnel watch failed: event object not of type v1.Pod")
|
||||
continue
|
||||
}
|
||||
ready := podutils.IsPodReady(pod)
|
||||
if pod.Spec.HostNetwork {
|
||||
for _, container := range pod.Spec.Containers {
|
||||
for _, port := range container.Ports {
|
||||
if port.Protocol == v1.ProtocolTCP {
|
||||
containerPort := fmt.Sprint(port.ContainerPort)
|
||||
if ready {
|
||||
logrus.Debugf("Tunnel authorizer adding Node Port %s", containerPort)
|
||||
a.ports[containerPort] = true
|
||||
} else {
|
||||
logrus.Debugf("Tunnel authorizer removing Node Port %s", containerPort)
|
||||
delete(a.ports, containerPort)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for _, ip := range pod.Status.PodIPs {
|
||||
if cidr, err := util.IPStringToIPNet(ip.IP); err == nil {
|
||||
if ready {
|
||||
logrus.Debugf("Tunnel authorizer adding Pod IP %s", cidr)
|
||||
a.cidrs.Insert(&podEntry{cidr: *cidr})
|
||||
} else {
|
||||
logrus.Debugf("Tunnel authorizer removing Pod IP %s", cidr)
|
||||
a.cidrs.Remove(*cidr)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// WatchEndpoints attempts to create tunnels to all supervisor addresses. Once the
|
||||
// apiserver is up, go into a watch loop, adding and removing tunnels as endpoints come
|
||||
// and go from the cluster.
|
||||
func (a *agentTunnel) watchEndpoints(ctx context.Context, apiServerReady <-chan struct{}, wg *sync.WaitGroup, tlsConfig *tls.Config, proxy proxy.Proxy) {
|
||||
// Attempt to connect to supervisors, storing their cancellation function for later when we
|
||||
// need to disconnect.
|
||||
disconnect := map[string]context.CancelFunc{}
|
||||
for _, address := range proxy.SupervisorAddresses() {
|
||||
if _, ok := disconnect[address]; !ok {
|
||||
disconnect[address] = a.connect(ctx, wg, address, tlsConfig)
|
||||
}
|
||||
}
|
||||
|
||||
<-apiServerReady
|
||||
endpoints := a.client.CoreV1().Endpoints(metav1.NamespaceDefault)
|
||||
fieldSelector := fields.Set{metav1.ObjectNameField: "kubernetes"}.String()
|
||||
lw := &cache.ListWatch{
|
||||
ListFunc: func(options metav1.ListOptions) (object runtime.Object, e error) {
|
||||
options.FieldSelector = fieldSelector
|
||||
return endpoints.List(ctx, options)
|
||||
},
|
||||
WatchFunc: func(options metav1.ListOptions) (i watch.Interface, e error) {
|
||||
options.FieldSelector = fieldSelector
|
||||
return endpoints.Watch(ctx, options)
|
||||
},
|
||||
}
|
||||
|
||||
_, _, watch, done := toolswatch.NewIndexerInformerWatcher(lw, &v1.Endpoints{})
|
||||
|
||||
defer func() {
|
||||
watch.Stop()
|
||||
<-done
|
||||
}()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case ev, ok := <-watch.ResultChan():
|
||||
endpoint, ok := ev.Object.(*v1.Endpoints)
|
||||
if !ok {
|
||||
logrus.Errorf("Tunnel watch failed: event object not of type v1.Endpoints")
|
||||
continue
|
||||
}
|
||||
|
||||
newAddresses := util.GetAddresses(endpoint)
|
||||
if reflect.DeepEqual(newAddresses, proxy.SupervisorAddresses()) {
|
||||
continue
|
||||
}
|
||||
proxy.Update(newAddresses)
|
||||
|
||||
validEndpoint := map[string]bool{}
|
||||
|
||||
for _, address := range proxy.SupervisorAddresses() {
|
||||
validEndpoint[address] = true
|
||||
if _, ok := disconnect[address]; !ok {
|
||||
disconnect[address] = a.connect(ctx, nil, address, tlsConfig)
|
||||
}
|
||||
}
|
||||
|
||||
for address, cancel := range disconnect {
|
||||
if !validEndpoint[address] {
|
||||
cancel()
|
||||
delete(disconnect, address)
|
||||
logrus.Infof("Stopped tunnel to %s", address)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// authorized determines whether or not a dial request is authorized.
|
||||
@@ -198,46 +306,24 @@ func (a *agentTunnel) authorized(ctx context.Context, proto, address string) boo
|
||||
logrus.Debugf("Tunnel authorizer checking dial request for %s", address)
|
||||
host, port, err := net.SplitHostPort(address)
|
||||
if err == nil {
|
||||
if proto == "tcp" && ports[port] && (host == "127.0.0.1" || host == "::1") {
|
||||
if proto == "tcp" && daemonconfig.KubeletReservedPorts[port] && (host == "127.0.0.1" || host == "::1") {
|
||||
return true
|
||||
}
|
||||
if ip := net.ParseIP(host); ip != nil {
|
||||
// lazy populate pod cidrs from the node object
|
||||
if a.mode == daemonconfig.EgressSelectorModePod && a.cidrs.Len() == 0 {
|
||||
if err := a.updatePodCIDRs(ctx); err != nil {
|
||||
logrus.Warnf("Tunnel authorizer failed to update CIDRs: %v", err)
|
||||
}
|
||||
}
|
||||
if nets, err := a.cidrs.ContainingNetworks(ip); err == nil && len(nets) > 0 {
|
||||
return true
|
||||
if p, ok := nets[0].(*podEntry); ok {
|
||||
if p.hostNet {
|
||||
return proto == "tcp" && a.ports[port]
|
||||
}
|
||||
return true
|
||||
}
|
||||
logrus.Debugf("Tunnel authorizer CIDR lookup returned unknown type for address %s", ip)
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func (a *agentTunnel) updatePodCIDRs(ctx context.Context) error {
|
||||
a.Lock()
|
||||
defer a.Unlock()
|
||||
// Return early if another goroutine updated the CIDRs while we were waiting to lock
|
||||
if a.cidrs.Len() != 0 {
|
||||
return nil
|
||||
}
|
||||
nodeName := os.Getenv("NODE_NAME")
|
||||
logrus.Debugf("Tunnel authorizer getting Pod CIDRs for %s", nodeName)
|
||||
node, err := a.client.CoreV1().Nodes().Get(ctx, nodeName, metav1.GetOptions{})
|
||||
if err != nil {
|
||||
return errors.Wrap(err, "failed to get Pod CIDRs")
|
||||
}
|
||||
for _, cidr := range node.Spec.PodCIDRs {
|
||||
if _, n, err := net.ParseCIDR(cidr); err == nil {
|
||||
logrus.Infof("Tunnel authorizer added Pod CIDR %s", cidr)
|
||||
a.cidrs.Insert(cidranger.NewBasicRangerEntry(*n))
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// connect initiates a connection to the remotedialer server. Incoming dial requests from
|
||||
// the server will be checked by the authorizer function prior to being fulfilled.
|
||||
func (a *agentTunnel) connect(rootCtx context.Context, waitGroup *sync.WaitGroup, address string, tlsConfig *tls.Config) context.CancelFunc {
|
||||
|
||||
Reference in New Issue
Block a user