基于遗传算法的Kubernetes Pod节点分配适配超资源场景需求
基于遗传算法的Kubernetes Pod节点分配方案优化(支持资源超配场景)
问题分析
你的现有实现仅能处理Pod总资源不超过节点总资源的场景,当资源超配时,虽然会统计未分配Pod,但遗传算法的个体编码、适应度逻辑没有针对"允许未分配"做针对性优化,容易出现分配逻辑不合理(比如强行尝试分配超资源的Pod)的情况。下面是针对性的优化方案和代码修改。
核心优化方向
- 扩展个体编码:新增
-1作为Pod未分配的状态标识,原有的0~n_nodes-1保留为节点索引 - 优化适应度函数:
- 保留资源利用率的奖励逻辑
- 调整未分配Pod的惩罚权重,使其与Pod的资源价值匹配,避免过度惩罚或惩罚不足
- 严格校验节点资源分配,超配的Pod视为未分配,不占用节点资源
- 适配遗传操作:交叉、突变逻辑需支持未分配状态的生成与传递
修改后的完整代码
from string import ascii_lowercase import numpy as np import random from itertools import compress import math import pandas as pd import random def create_pods_and_nodes(n_pods=40, n_nodes=15): # Create pod and node names pod = ['pod_' + str(i+1) for i in range(n_pods)] node = ['node_' + str(i+1) for i in range(n_nodes)] # Define CPU and RAM options cpu = [2**i for i in range(1, 8)] # 2, 4, 8, 16, 32, 64, 128 ram = [2**i for i in range(2, 10)] # 4, 8, 16, ..., 8192 # Create the pods DataFrame pods = pd.DataFrame({ 'pod': pod, 'cpu': random.choices(cpu[0:3], k=n_pods), # Small CPU for pods 'ram': random.choices(ram[0:4], k=n_pods), # Small RAM for pods }) # Create the nodes DataFrame nodes = pd.DataFrame({ 'node': node, 'cpu': random.choices(cpu[4:len(cpu)-1], k=n_nodes), # Larger CPU for nodes 'ram': random.choices(ram[4:len(ram)-1], k=n_nodes), # Larger RAM for nodes }) return pods, nodes # Example usage pods, nodes = create_pods_and_nodes(n_pods=46, n_nodes=6) # Display the results print("Pods DataFrame:\n", pods.head()) print("\nNodes DataFrame:\n", nodes.head()) print(f"total CPU pods: {np.sum(pods['cpu'])}") print(f"total RAM pods: {np.sum(pods['ram'])}") print('\n') print(f"total CPU nodes: {np.sum(nodes['cpu'])}") print(f"total RAM nodes: {np.sum(nodes['ram'])}") # Genetic Algorithm Parameters POPULATION_SIZE = 100 GENERATIONS = 50 MUTATION_RATE = 0.1 TOURNAMENT_SIZE = 5 # 惩罚系数:基于Pod平均资源价值设置,让未分配的损失与浪费资源的损失匹配 AVG_POD_RESOURCE = np.mean(pods['cpu'] + pods['ram']) PUNISHMENT_WEIGHT = AVG_POD_RESOURCE def create_individual(): # 新增-1表示Pod未分配 return [random.choice([-1] + list(range(len(nodes)))) for _ in range(len(pods))] def create_population(size): return [create_individual() for _ in range(size)] def fitness(individual): total_cpu_used = np.zeros(len(nodes)) total_ram_used = np.zeros(len(nodes)) unallocated_pods = 0 for pod_idx, node_idx in enumerate(individual): pod_cpu = pods.iloc[pod_idx]['cpu'] pod_ram = pods.iloc[pod_idx]['ram'] if node_idx == -1: # 标记为未分配 unallocated_pods += 1 else: # 检查节点资源是否足够 if total_cpu_used[node_idx] + pod_cpu <= nodes.iloc[node_idx]['cpu'] and total_ram_used[node_idx] + pod_ram <= nodes.iloc[node_idx]['ram']: total_cpu_used[node_idx] += pod_cpu total_ram_used[node_idx] += pod_ram else: # 分配失败,视为未分配 unallocated_pods += 1 # 适应度:总资源利用率 - 未分配Pod的惩罚 return (total_cpu_used.sum() + total_ram_used.sum()) - (unallocated_pods * PUNISHMENT_WEIGHT) def select(population): tournament = random.sample(population, TOURNAMENT_SIZE) return max(tournament, key=fitness) def crossover(parent1, parent2): crossover_point = random.randint(1, len(pods) - 1) child1 = parent1[:crossover_point] + parent2[crossover_point:] child2 = parent2[:crossover_point] + parent1[crossover_point:] return child1, child2 def mutate(individual): for idx in range(len(individual)): if random.random() < MUTATION_RATE: # 突变时可以选择未分配或任意节点 individual[idx] = random.choice([-1] + list(range(len(nodes)))) def genetic_algorithm(): population = create_population(POPULATION_SIZE) # 精英保留:每代保留最优个体,避免退化 best_individual_ever = max(population, key=fitness) for generation in range(GENERATIONS): new_population = [] for _ in range(POPULATION_SIZE // 2): parent1 = select(population) parent2 = select(population) child1, child2 = crossover(parent1, parent2) mutate(child1) mutate(child2) new_population.extend([child1, child2]) # 加入精英个体 new_population.append(best_individual_ever) # 保持种群大小,移除随机一个个体 new_population.pop(random.randint(0, len(new_population)-1)) population = new_population # 更新全局最优个体 current_best = max(population, key=fitness) if fitness(current_best) > fitness(best_individual_ever): best_individual_ever = current_best # Print the best fitness of this generation print(f"Generation {generation + 1}: Best Fitness = {fitness(best_individual_ever)}") # Return the best individual found return best_individual_ever # Run the genetic algorithm print("Starting Genetic Algorithm...") best_allocation = genetic_algorithm() print("Genetic Algorithm completed.\n") # Create the allocation DataFrame allocation_df = pd.DataFrame({ 'Pod': pods['pod'], 'Node': [nodes.iloc[node_idx]['node'] if node_idx != -1 else '未分配' for node_idx in best_allocation], 'Pod_Resources': [list(pods.iloc[i][['cpu', 'ram']]) for i in range(len(best_allocation))], 'Node_Resources': [list(nodes.iloc[node_idx][['cpu', 'ram']]) if node_idx != -1 else [0, 0] for node_idx in best_allocation] }) # Print the allocation DataFrame print("\nAllocation DataFrame:") print(allocation_df) # Summarize total CPU and RAM utilization for each node # 过滤未分配的Pod,只统计已分配的 allocated_df = allocation_df[allocation_df['Node'] != '未分配'] node_utilization_df = allocated_df.groupby('Node').agg( Total_CPU_Used=pd.NamedAgg(column='Pod_Resources', aggfunc=lambda x: sum([res[0] for res in x])), Total_RAM_Used=pd.NamedAgg(column='Pod_Resources', aggfunc=lambda x: sum([res[1] for res in x])), Node_CPU=pd.NamedAgg(column='Node_Resources', aggfunc=lambda x: x.iloc[0][0]), Node_RAM=pd.NamedAgg(column='Node_Resources', aggfunc=lambda x: x.iloc[0][1]) ) # Calculate CPU and RAM utilization percentages for each node node_utilization_df['CPU_Utilization'] = (node_utilization_df['Total_CPU_Used'] / node_utilization_df['Node_CPU']) * 100 node_utilization_df['RAM_Utilization'] = (node_utilization_df['Total_RAM_Used'] / node_utilization_df['Node_RAM']) * 100 # Print the total CPU and RAM utilization for each node print("\nTotal CPU and RAM utilization for each node:") print(node_utilization_df) # 统计未分配Pod unallocated_count = len(allocation_df[allocation_df['Node'] == '未分配']) print(f"\n未分配Pod数量: {unallocated_count}")
额外改进建议
- 动态惩罚系数:针对不同资源量级的Pod设置不同惩罚权重,比如大资源Pod未分配时惩罚更重,优先保障大资源Pod的分配
- 资源均衡性优化:在适应度函数中加入节点资源利用率的方差惩罚项,避免出现部分节点满载、部分节点闲置的情况
- 提前终止机制:当连续N代(比如10代)适应度没有提升时,提前结束算法,减少不必要的计算
- 并行计算适配:将适应度计算逻辑改为并行执行,提升大规模Pod/节点场景下的运行效率
- Pod优先级支持:如果业务场景中有Pod优先级,可在适应度函数中加入优先级权重,优先分配高优先级Pod
内容的提问来源于stack exchange,提问作者AndCh
相关产品推荐
相关产品推荐

