使用Bash或Python合并a.txt与b.txt生成c.txt的技术求助
Bash 实现方案
使用 awk 可高效完成文本合并逻辑,以下是完整脚本:
#!/bin/bash awk ' BEGIN { FS = "|" OFS = "|" header_done = 0 delete b_rules } # 先处理b.txt,提取头部信息与规则 FILENAME == "b.txt" { if ($0 ~ /^# role version :/) { split($0, parts, ": ") b_version_val = parts[2] } else if ($0 ~ /^# security ID/) { rule_header = $0 } else if ($0 ~ /^# rule_/) { gsub(/^[ \t]+|[ \t]+$/, "", $1) sec_id = $1 gsub(/^[ \t]+|[ \t]+$/, "", $3) if (length($3) > 0) { b_rules[sec_id] = $0 b_has_id[sec_id] = 1 } else { b_has_id[sec_id] = 1 } } next } # 处理a.txt,生成c.txt主体内容 FILENAME == "a.txt" { if (!header_done) { if ($0 ~ /^# date :/) { print $0 " => date from A file" } else if ($0 ~ /^# profile name :/) { print $0 " => zone from A file " } else if ($0 ~ /^# role version :/) { printf "# role version : %s => version from B file\n", b_version_val } else if ($0 ~ /^# security ID/) { print rule_header header_done = 1 } else { print $0 } } else { if ($0 ~ /^# rule_/) { gsub(/^[ \t]+|[ \t]+$/, "", $1) sec_id = $1 if (sec_id in b_rules) { print b_rules[sec_id] processed_ids[sec_id] = 1 } else { print $0 } } } next } # 追加b.txt中独有的、custom列非空的规则 END { for (sec_id in b_has_id) { if (!(sec_id in processed_ids) && sec_id in b_rules) { print b_rules[sec_id] } } } ' b.txt a.txt > c.txt
核心逻辑说明
- 优先读取
b.txt,提取版本信息并将custom列非空的规则存入字典,用于后续替换 - 读取
a.txt时,替换头部的role version为b.txt的版本并添加来源注释;规则行若在b.txt中有对应且custom非空,则用b.txt的行覆盖 - 最后补充
b.txt中独有的、符合条件的规则
Python 实现方案
Python实现逻辑更直观,便于后续扩展复杂需求:
def parse_file(file_path): """解析文件,返回头部列表、规则字典和规则表头""" header = [] rules = {} rule_header = "" with open(file_path, 'r') as f: for line in f: line = line.rstrip('\n') if line.startswith('# Ansible managed') or line.startswith('#########################################################'): header.append(line) elif line in ('#', ''): header.append(line) elif line.startswith('# date :') or line.startswith('# profile name :'): header.append(line) elif line.startswith('# role version :'): header.append(line) elif line.startswith('# security ID'): rule_header = line header.append(line) elif line.startswith('# rule_'): parts = [p.strip() for p in line.split('|')] sec_id = parts[0].lstrip('# ') rules[sec_id] = { 'line': line, 'custom': parts[2] } return header, rules, rule_header # 解析两个输入文件 a_header, a_rules, _ = parse_file('a.txt') b_header, b_rules, _ = parse_file('b.txt') # 构建新头部 new_header = [] b_version = "" # 先提取b的版本值 for line in b_header: if line.startswith('# role version :'): b_version = line.split(': ')[1] break # 替换头部内容 for line in a_header: if line.startswith('# date :'): new_header.append(f"{line} => date from A file") elif line.startswith('# profile name :'): new_header.append(f"{line} => zone from A file ") elif line.startswith('# role version :'): new_header.append(f"# role version : {b_version} => version from B file") else: new_header.append(line) # 合并规则 merged_rules = a_rules.copy() # 用b中custom非空的规则覆盖 for sec_id, b_rule in b_rules.items(): if b_rule['custom'] != '': merged_rules[sec_id] = b_rule # 生成c.txt with open('c.txt', 'w') as f: # 写入头部 for line in new_header: f.write(f"{line}\n") # 写入a中原有规则(已完成覆盖) for sec_id in a_rules: f.write(f"{merged_rules[sec_id]['line']}\n") # 写入b中独有的、custom非空的规则 for sec_id in b_rules: if sec_id not in a_rules and b_rules[sec_id]['custom'] != '': f.write(f"{b_rules[sec_id]['line']}\n")
核心逻辑说明
parse_file函数统一解析文件,将规则按security ID存入字典,便于后续查找替换- 构建新头部时,替换
role version为b.txt的版本并添加来源注释 - 合并规则时,先保留
a.txt的所有规则,再用b.txt中custom列非空的规则覆盖对应条目,最后补充b.txt独有的符合条件的规则 - 写入文件时保持
a.txt原有规则的顺序,再追加新规则
内容的提问来源于stack exchange,提问作者Dany Seb
相关产品推荐
相关产品推荐

