You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Kubernetes部署Flower联邦学习服务器遭遇端口占用问题

Kubernetes部署Flower联邦学习服务器端口冲突问题解决

我在Kubernetes集群中使用自定义镜像fl-server:latest部署Flower联邦学习服务器,要求启动前自动检测并使用可用端口以避免冲突。但部署后日志持续报错目标端口已被占用,尽管在容器内通过ss/netstat验证无其他服务占用该端口。


现有实现代码

1. Kubernetes Deployment与Service创建代码

from kubernetes import client, config
from uuid import uuid4

class DeploymentFL:
    
    def __init__(self):
        pass

    def create_deployment_object(self, model_name: str, port_k8s: str, port_flower: str, model_uuid: str, host: str = "192.168.49.2") -> client.V1Deployment:
        """
        Create a Kubernetes deployment object for a Federated Learning server.
        
        Args:
            model_name (str): The name of the model to be used.
            port_k8s (str): The port number for the Kubernetes service.
            port_flower (str): The port number for the Flower server.
            model_uuid (str): The uuid of the model to identify it.
            host (str): The host address of the server.

        Return:
            client.V1Deployment: The deployment object for the Kubernetes API.
        """
        try:
            sanitized_model_name = f"{model_name.lower().replace('_', '-')}"
            name_container = f"fl-server-{sanitized_model_name}-{model_uuid}"
            
            # Define the container with environment variables and health checks
            container = client.V1Container(
                name=name_container,
                image="fl-server:latest",
                image_pull_policy="Never",
                ports=[
                    client.V1ContainerPort(container_port=int(port_k8s)),
                ],
                env=[
                    client.V1EnvVar(name="FL_HOST", value=host),
                    client.V1EnvVar(name="FL_PORT", value=str(port_k8s)),
                ],
                volume_mounts=[client.V1VolumeMount(
                    name='model-storage',
                    mount_path='/mnt/data'
                )],
                liveness_probe=client.V1Probe(
                    _exec=client.V1ExecAction(command=["/bin/sh", "-c", f"nc -zv {host} {port_k8s}"]),
                    initial_delay_seconds=60,
                    period_seconds=10,
                    timeout_seconds=5,
                    failure_threshold=3,
                ),
                readiness_probe=client.V1Probe(
                    _exec=client.V1ExecAction(command=["/bin/sh", "-c", f"nc -zv {host} {port_k8s}"]),
                    initial_delay_seconds=30,
                    period_seconds=10,
                    timeout_seconds=5,
                    failure_threshold=3,
                )
            )
            
            # Define the Pod template
            template = client.V1PodTemplateSpec(
                metadata=client.V1ObjectMeta(labels={"app": name_container}),
                spec=client.V1PodSpec(containers=[container], volumes=[client.V1Volume(
                    name='model-storage',
                    persistent_volume_claim=client.V1PersistentVolumeClaimVolumeSource(claim_name='model-pvc')
                )])
            )
            
            # Define the Deployment specification
            spec = client.V1DeploymentSpec(
                replicas=1,
                template=template,
                selector={'matchLabels': {'app': name_container}}
            )
            
            name_deployment = f"{name_container}-deployment"
            deployment = client.V1Deployment(
                metadata=client.V1ObjectMeta(name=name_deployment),
                spec=spec
            )
            
            return deployment
        except Exception as e:
            print(f"Error at create deployment object: {e}")

    def create_service_object(self, model_name: str, port: int, model_uuid: str) -> client.V1Service:
        """
        Create a Kubernetes service object for a Federated Learning server.

        Args:
            model_name (str): The name of the model to be used.
            port (int): The port number for the server.
            model_uuid (str): The uuid of the model to identify it.

        Return:
            client.V1Service: The service object for the Kubernetes API.
        """
        try:
            sanitized_model_name = model_name.lower().replace("_", "-")
            service_name = f"fl-server-{sanitized_model_name}-{model_uuid}-service"
            service = client.V1Service(
                metadata=client.V1ObjectMeta(name=service_name),
                spec=client.V1ServiceSpec(
                    selector={"app": f"fl-server-{sanitized_model_name}-{model_uuid}"},
                    ports=[client.V1ServicePort(port=port, target_port=port)],
                    type="NodePort"
                )
            )
            return service, service_name
        except Exception as e:
            print(f"Error at create service object: {e}")

2. Flower服务器镜像脚本(fl_server.py)

import flwr as fl
import tensorflow as tf
import os
import socket

def is_port_in_use(host, port):
    """Check if a port is in use."""
    with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
        return s.connect_ex((host, port)) == 0

def run_server(host: str, port: int):
    """
    Start a Federated Learning server with the given model.

    Args:
        host (str): The host address for the server.
        port (int): The port number for the server.

    Return:
        None
    """
    try:
        # Define the Federated Learning strategy
        strategy = fl.server.strategy.FedAvg(
            fraction_fit=0.1,
            fraction_evaluate=0.1,
            min_fit_clients=2,
            min_evaluate_clients=2,
            min_available_clients=2,
        )

        # Start the Flower server
        fl.server.start_server(
            server_address=f"{host}:{port}",
            config=fl.server.ServerConfig(num_rounds=3),
            strategy=strategy,
        )
    except Exception as e:
        print(f"Error starting fl server: {e}")

if __name__ == "__main__":
    # Read environment variables
    host = os.getenv("FL_HOST", "192.168.49.2")
    port = int(os.getenv("FL_PORT", "30878"))

    # Find an available port
    original_port = port
    while is_port_in_use(host, port):
        print(f"Port {port} is already in use. Trying another port...")
        port += 1

    print(f"Starting Flower server on {host}:{port} (initially tried {original_port})")
    run_server(host, port)

3. Dockerfile

FROM python:3.10

# Install dependencies
COPY requirements.txt .
RUN pip install -r requirements.txt

# Copy the application code
COPY fl_server.py /app/fl_server.py
WORKDIR /app

# Run the server
CMD ["python", "fl_server.py"]

已执行的排查步骤

  • 本地环境验证端口检测逻辑正常,能正确识别占用的端口并切换
  • 确保Deployment使用的是最新构建的镜像
  • 确认Service选择器与Pod标签完全匹配
  • 进入容器执行ss -tulpn和netstat -tulpn,未发现目标端口被其他进程占用

问题分析

日志报错显示端口已被占用,但容器内无实际占用,核心原因有三个:

  1. 服务器绑定地址错误:
    当前Flower服务器尝试绑定Node的IP(192.168.49.2),但Kubernetes Pod默认使用桥接网络,容器内无法直接绑定主机的IP地址。此时Flower启动时会因无法绑定指定IP而报错,错误信息被误判为端口占用。

  2. 端口检测逻辑不适用容器环境:
    is_port_in_use函数通过连接Node IP:端口来判断是否占用,但这仅能检测Node上的端口是否被其他服务监听,无法判断容器内是否能绑定该端口。本地环境中脚本直接运行在主机上,连接和绑定的是同一个地址,所以逻辑有效;但容器内连接Node IP的端口结果,和容器自身能否绑定该端口无关。

  3. 健康检查逻辑错误:
    存活/就绪探针使用nc -zv {host} {port_k8s}检测Node IP的端口,而此时Flower服务器根本无法绑定该IP,导致探针持续失败。


解决方案

方案1:修改Flower服务器绑定地址与端口检测逻辑

将服务器绑定地址改为0.0.0.0(容器内所有接口),同时调整端口检测逻辑为尝试绑定容器内端口,而非连接外部地址:

修改fl_server.py:

def is_port_in_use(port):
    """Check if port is available to bind inside container"""
    with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
        try:
            s.bind(('0.0.0.0', port))
            return False
        except socket.error:
            return True

if __name__ == "__main__":
    # 服务器绑定容器内的0.0.0.0,客户端通过Node IP:NodePort访问
    host = "0.0.0.0"
    port = int(os.getenv("FL_PORT", "30878"))

    # Find an available port
    original_port = port
    while is_port_in_use(port):
        print(f"Port {port} is already in use inside container. Trying another port...")
        port += 1

    print(f"Starting Flower server on {host}:{port} (initially tried {original_port})")
    run_server(host, port)

方案2:调整Deployment配置

  • 修改环境变量FL_HOST为0.0.0.0,确保容器内的Flower服务器绑定正确地址
  • 修正健康检查逻辑,改为检测容器内的本地端口:
# 在create_deployment_object函数中修改liveness_probe和readiness_probe
liveness_probe=client.V1Probe(
    _exec=client.V1ExecAction(command=["/bin/sh", "-c", f"nc -zv localhost {port_k8s}"]),
    initial_delay_seconds=60,
    period_seconds=10,
    timeout_seconds=5,
    failure_threshold=3,
),
readiness_probe=client.V1Probe(
    _exec=client.V1ExecAction(command=["/bin/sh", "-c", f"nc -zv localhost {port_k8s}"]),
    initial_delay_seconds=30,
    period_seconds=10,
    timeout_seconds=5,
    failure_threshold=3,
)

方案3:适配Kubernetes端口管理(可选)

如果需要固定NodePort,建议由Kubernetes自动分配端口(或指定端口范围),而非在容器内动态切换端口。因为容器内动态切换端口后,Service的targetPort无法同步更新,会导致流量无法转发。此时可以去掉容器内的端口检测逻辑,直接使用Deployment中指定的端口,由Kubernetes负责端口冲突管理。


验证步骤

  1. 重新构建镜像:docker build -t fl-server:latest .
  2. 更新Deployment:确保新镜像被使用
  3. 查看Pod日志:确认Flower服务器正常启动,无端口占用报错
  4. 测试服务连通性:通过Node IP:NodePort访问Flower服务器,验证客户端能正常连接

内容的提问来源于stack exchange,提问作者tobeal

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.17 10:29:56