You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于遗传算法的Kubernetes Pod节点分配适配超资源场景需求

基于遗传算法的Kubernetes Pod节点分配方案优化(支持资源超配场景)

问题分析

你的现有实现仅能处理Pod总资源不超过节点总资源的场景,当资源超配时,虽然会统计未分配Pod,但遗传算法的个体编码、适应度逻辑没有针对"允许未分配"做针对性优化,容易出现分配逻辑不合理(比如强行尝试分配超资源的Pod)的情况。下面是针对性的优化方案和代码修改。

核心优化方向

  1. 扩展个体编码:新增-1作为Pod未分配的状态标识,原有的0~n_nodes-1保留为节点索引
  2. 优化适应度函数:
    • 保留资源利用率的奖励逻辑
    • 调整未分配Pod的惩罚权重,使其与Pod的资源价值匹配,避免过度惩罚或惩罚不足
    • 严格校验节点资源分配,超配的Pod视为未分配,不占用节点资源
  3. 适配遗传操作:交叉、突变逻辑需支持未分配状态的生成与传递

修改后的完整代码

from string import ascii_lowercase
import numpy as np
import random
from itertools import compress
import math
import pandas as pd
import random

def create_pods_and_nodes(n_pods=40, n_nodes=15):
    # Create pod and node names
    pod = ['pod_' + str(i+1) for i in range(n_pods)]
    node = ['node_' + str(i+1) for i in range(n_nodes)]

    # Define CPU and RAM options
    cpu = [2**i for i in range(1, 8)]  # 2, 4, 8, 16, 32, 64, 128
    ram = [2**i for i in range(2, 10)]  # 4, 8, 16, ..., 8192

    # Create the pods DataFrame
    pods = pd.DataFrame({
        'pod': pod,
        'cpu': random.choices(cpu[0:3], k=n_pods),  # Small CPU for pods
        'ram': random.choices(ram[0:4], k=n_pods),  # Small RAM for pods
    })

    # Create the nodes DataFrame
    nodes = pd.DataFrame({
        'node': node,
        'cpu': random.choices(cpu[4:len(cpu)-1], k=n_nodes),  # Larger CPU for nodes
        'ram': random.choices(ram[4:len(ram)-1], k=n_nodes),  # Larger RAM for nodes
    })

    return pods, nodes

# Example usage
pods, nodes = create_pods_and_nodes(n_pods=46, n_nodes=6)

# Display the results
print("Pods DataFrame:\n", pods.head())
print("\nNodes DataFrame:\n", nodes.head())

print(f"total CPU pods: {np.sum(pods['cpu'])}")
print(f"total RAM pods: {np.sum(pods['ram'])}")
print('\n')
print(f"total CPU nodes: {np.sum(nodes['cpu'])}")
print(f"total RAM nodes: {np.sum(nodes['ram'])}")

# Genetic Algorithm Parameters
POPULATION_SIZE = 100
GENERATIONS = 50
MUTATION_RATE = 0.1
TOURNAMENT_SIZE = 5
# 惩罚系数:基于Pod平均资源价值设置,让未分配的损失与浪费资源的损失匹配
AVG_POD_RESOURCE = np.mean(pods['cpu'] + pods['ram'])
PUNISHMENT_WEIGHT = AVG_POD_RESOURCE

def create_individual():
    # 新增-1表示Pod未分配
    return [random.choice([-1] + list(range(len(nodes)))) for _ in range(len(pods))]

def create_population(size):
    return [create_individual() for _ in range(size)]

def fitness(individual):
    total_cpu_used = np.zeros(len(nodes))
    total_ram_used = np.zeros(len(nodes))
    unallocated_pods = 0

    for pod_idx, node_idx in enumerate(individual):
        pod_cpu = pods.iloc[pod_idx]['cpu']
        pod_ram = pods.iloc[pod_idx]['ram']

        if node_idx == -1:
            # 标记为未分配
            unallocated_pods += 1
        else:
            # 检查节点资源是否足够
            if total_cpu_used[node_idx] + pod_cpu <= nodes.iloc[node_idx]['cpu'] and total_ram_used[node_idx] + pod_ram <= nodes.iloc[node_idx]['ram']:
                total_cpu_used[node_idx] += pod_cpu
                total_ram_used[node_idx] += pod_ram
            else:
                # 分配失败,视为未分配
                unallocated_pods += 1

    # 适应度:总资源利用率 - 未分配Pod的惩罚
    return (total_cpu_used.sum() + total_ram_used.sum()) - (unallocated_pods * PUNISHMENT_WEIGHT)

def select(population):
    tournament = random.sample(population, TOURNAMENT_SIZE)
    return max(tournament, key=fitness)

def crossover(parent1, parent2):
    crossover_point = random.randint(1, len(pods) - 1)
    child1 = parent1[:crossover_point] + parent2[crossover_point:]
    child2 = parent2[:crossover_point] + parent1[crossover_point:]
    return child1, child2

def mutate(individual):
    for idx in range(len(individual)):
        if random.random() < MUTATION_RATE:
            # 突变时可以选择未分配或任意节点
            individual[idx] = random.choice([-1] + list(range(len(nodes))))

def genetic_algorithm():
    population = create_population(POPULATION_SIZE)
    
    # 精英保留:每代保留最优个体,避免退化
    best_individual_ever = max(population, key=fitness)
    
    for generation in range(GENERATIONS):
        new_population = []
        for _ in range(POPULATION_SIZE // 2):
            parent1 = select(population)
            parent2 = select(population)
            child1, child2 = crossover(parent1, parent2)
            mutate(child1)
            mutate(child2)
            new_population.extend([child1, child2])
        
        # 加入精英个体
        new_population.append(best_individual_ever)
        # 保持种群大小,移除随机一个个体
        new_population.pop(random.randint(0, len(new_population)-1))
        
        population = new_population
        
        # 更新全局最优个体
        current_best = max(population, key=fitness)
        if fitness(current_best) > fitness(best_individual_ever):
            best_individual_ever = current_best

        # Print the best fitness of this generation
        print(f"Generation {generation + 1}: Best Fitness = {fitness(best_individual_ever)}")

    # Return the best individual found
    return best_individual_ever

# Run the genetic algorithm
print("Starting Genetic Algorithm...")
best_allocation = genetic_algorithm()
print("Genetic Algorithm completed.\n")

# Create the allocation DataFrame
allocation_df = pd.DataFrame({
    'Pod': pods['pod'],
    'Node': [nodes.iloc[node_idx]['node'] if node_idx != -1 else '未分配' for node_idx in best_allocation],
    'Pod_Resources': [list(pods.iloc[i][['cpu', 'ram']]) for i in range(len(best_allocation))],
    'Node_Resources': [list(nodes.iloc[node_idx][['cpu', 'ram']]) if node_idx != -1 else [0, 0] for node_idx in best_allocation]
})

# Print the allocation DataFrame
print("\nAllocation DataFrame:")
print(allocation_df)

# Summarize total CPU and RAM utilization for each node
# 过滤未分配的Pod,只统计已分配的
allocated_df = allocation_df[allocation_df['Node'] != '未分配']
node_utilization_df = allocated_df.groupby('Node').agg(
    Total_CPU_Used=pd.NamedAgg(column='Pod_Resources', aggfunc=lambda x: sum([res[0] for res in x])),
    Total_RAM_Used=pd.NamedAgg(column='Pod_Resources', aggfunc=lambda x: sum([res[1] for res in x])),
    Node_CPU=pd.NamedAgg(column='Node_Resources', aggfunc=lambda x: x.iloc[0][0]),
    Node_RAM=pd.NamedAgg(column='Node_Resources', aggfunc=lambda x: x.iloc[0][1])
)

# Calculate CPU and RAM utilization percentages for each node
node_utilization_df['CPU_Utilization'] = (node_utilization_df['Total_CPU_Used'] / node_utilization_df['Node_CPU']) * 100
node_utilization_df['RAM_Utilization'] = (node_utilization_df['Total_RAM_Used'] / node_utilization_df['Node_RAM']) * 100

# Print the total CPU and RAM utilization for each node
print("\nTotal CPU and RAM utilization for each node:")
print(node_utilization_df)

# 统计未分配Pod
unallocated_count = len(allocation_df[allocation_df['Node'] == '未分配'])
print(f"\n未分配Pod数量: {unallocated_count}")

额外改进建议

  • 动态惩罚系数:针对不同资源量级的Pod设置不同惩罚权重,比如大资源Pod未分配时惩罚更重,优先保障大资源Pod的分配
  • 资源均衡性优化:在适应度函数中加入节点资源利用率的方差惩罚项,避免出现部分节点满载、部分节点闲置的情况
  • 提前终止机制:当连续N代(比如10代)适应度没有提升时,提前结束算法,减少不必要的计算
  • 并行计算适配:将适应度计算逻辑改为并行执行,提升大规模Pod/节点场景下的运行效率
  • Pod优先级支持:如果业务场景中有Pod优先级,可在适应度函数中加入优先级权重,优先分配高优先级Pod

内容的提问来源于stack exchange,提问作者AndCh

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.17 18:24:54