如何用Python自动拆分FORTRAN 77文件中超长行?
解决FORTRAN 77超长行自动拆分问题
需求说明
处理遗留FORTRAN 77文件时,需解决部分行超出72字符限制的问题。已编写Python脚本检测超长行,现需扩展脚本实现按规则自动拆分,拆分后的子行需在第6列以&开头。
拆分规则
- 若超长部分位于单/双引号包裹的字符串内,精确在第72字符处拆分
- 若属于逗号分隔的变量名列表,在第72字符前最后一个逗号处拆分
- 若为算术表达式,在第72字符前最后一个算术运算符处拆分
现有检测脚本
import sys def show_long_lines(filename): """Display lines from the file longer than 72 characters.""" try: with open(filename, "r") as file: for line_number, line in enumerate(file, start=1): if len(line) > 72 and line.lstrip()[0] not in "cC*!": print(f"{line_number}: {line}") except FileNotFoundError: print("File not found. Please check the file path and try again.") except Exception as e: print(f"An error occurred: {e}") if __name__ == "__main__": if len(sys.argv) != 2: print("Usage: python script.py filename") else: filename = sys.argv[1] show_long_lines(filename)
超长行示例
WRITE(6,*)' COEFFICIENTS HAVE 20 DIGITS, UROUND=',WORK(1) WRITE(6,*)' CURIOUS INPUT FOR IWORK(5,6,7)=',NIND1,NIND2,NIND3 WRITE (6,*) 'BANDWITH OF "MAS" NOT SMALLER THAN BANDWITH OF WRITE(6,*)' HESSENBERG OPTION ONLY FOR EXPLICIT EQUATIONS WITH & JAC,IJAC,MLJAC,MUJAC,JACLAG,MAS,MLMAS,MUMAS,SOLOUT,IOUT,IDID, XLAG(1,IL)=ARGLAG(IL,X1,ZL,RPAR,IPAR,PHI,PAST,IPAST,NRDS, XLAG(2,IL)=ARGLAG(IL,X2,ZL(N+1),RPAR,IPAR,PHI,PAST,IPAST,NRDS, XLAG(3,IL)=ARGLAG(IL,X3,ZL(N2+1),RPAR,IPAR,PHI,PAST,IPAST,NRDS, IF (ICOUN(1,IL)+ICOUN(2,IL)+ICOUN(3,IL).GE.1) CALJACL=.TRUE. & ' WARNING!: ADVANCED ARGUMENTS ARE USED AT X= ',XACT & (13.D0-7.D0*Sqrt(6.D0)+5.D0*(-2.D0+3.D0*Sqrt(6.D0))*S2) FJACL(I1+N,J1+2*N)=FJACL(I1+N,J1+2*N)+AI23H*FMAS(I1,J1) FJACL(I1+2*N,J1+N)=FJACL(I1+2*N,J1+N)+AI32H*FMAS(I1,J1) FJACL(I1+2*N,J1+2*N)=FJACL(I1+2*N,J1+2*N)+AI33H*FMAS(I1,J1) IF (ERR.GE.1.D0.OR.((ERR/ERRACC.GE.TCKBP).AND.(.NOT.BPD))) THEN WRITE(6,*) 'Found a BP at ', X, ', decrementing IGRID.' Z2(N-NDIMN+I)=F2(IPAST(NRDS+I))+HE*F2(N-NDIMN+I) Z3(N-NDIMN+I)=F3(IPAST(NRDS+I))+HE*F3(N-NDIMN+I) XL =ARGLAG(ILBP,X+HE,Z2,RPAR,IPAR,PHI,PAST,IPAST,NRDS, XLR=ARGLAG(ILBP,X+HE,Z3,RPAR,IPAR,PHI,PAST,IPAST,NRDS, & ' WARNING!: SOLUTION DOES NOT EXIST AT X= ',X & ' WARNING!: SOLUTION IS NOT UNIQUE AT X= ',X IF (THETA.LE.THET.AND.QT.GE.QUOT1.AND.QT.LE.QUOT2) THEN 10 THETA=(XLAG-(PAST(IPOS)+PAST(IPOS+IDIF-1)))/PAST(IPOS+IDIF-1) YLAGR5=PAST(I)+THETA*(PAST(NRDS+I)+(THETA-C2M1)*(PAST(2*NRDS+I) ALN = ARGLAG(IL,XA,YADV,RPAR,IPAR,PHI,PAST,IPAST,NRDS,
实现方案
核心逻辑
对每个非注释的超长行,按字符串场景→逗号列表场景→算术表达式场景的优先级判断类型,再执行对应拆分逻辑。拆分后生成的子行需保持原缩进格式,且在第6列(索引5,从0开始)添加&。
场景识别与拆分实现
字符串场景处理
- 遍历行的前72个字符,跟踪引号状态:记录当前是否处于单引号或双引号包裹的字符串内(FORTRAN中两个连续同类型引号表示转义,比如
''是字符串内的单引号)。 - 若第72字符处于引号内,直接在第72位拆分:前半行末尾加
&,后半行保留原缩进,在第6列插入&后接剩余内容。
- 遍历行的前72个字符,跟踪引号状态:记录当前是否处于单引号或双引号包裹的字符串内(FORTRAN中两个连续同类型引号表示转义,比如
逗号分隔列表场景处理
- 确认不在字符串内后,在0-71范围内从后往前查找最后一个逗号的位置。
- 在该逗号处拆分:前半行到逗号结束,末尾加
&;后半行在第6列加&,然后接逗号后的内容(保留原空格)。
算术表达式场景处理
- 定义FORTRAN常用算术/逻辑运算符集合:
['=', '+', '-', '*', '/', '.GE.', '.LE.', '.GT.', '.LT.', '.EQ.', '.NE.', '.AND.', '.OR.', '.NOT.'],优先匹配多字符运算符。 - 在0-71范围内从后往前查找最后一个完整运算符的结束位置,确保运算符不是变量名的一部分。
- 在运算符后拆分:前半行到运算符结束,末尾加
&;后半行在第6列加&,接运算符后的内容。
- 定义FORTRAN常用算术/逻辑运算符集合:
修改后的完整脚本
import sys def is_comment_line(line): stripped = line.lstrip() return len(stripped) == 0 or stripped[0] in "cC*!" def find_quote_end(line, start_idx): """Find the end of a quoted string starting at start_idx, handling escaped quotes.""" quote_char = line[start_idx] idx = start_idx + 1 while idx < len(line): if line[idx] == quote_char: # Check for escaped quote (two consecutive same quotes) if idx + 1 < len(line) and line[idx + 1] == quote_char: idx += 2 continue return idx idx += 1 return len(line) # Unclosed quote, treat as end of line def split_string_line(line): """Split line at column 72 if inside a string.""" max_col = 72 # Check if we're inside a string at max_col in_string = False current_quote = None idx = 0 while idx < max_col: if line[idx] in "'\"": if not in_string: in_string = True current_quote = line[idx] idx = find_quote_end(line, idx) else: if line[idx] == current_quote: # Check for escaped quote if idx + 1 < max_col and line[idx + 1] == current_quote: idx += 2 continue in_string = False current_quote = None idx += 1 if in_string: # Split exactly at 72 first_part = line[:72].rstrip() + "&\n" # Create continuation line: 5 spaces + & + remaining content indent = line[:line.find(line.lstrip())] cont_indent = " " * 5 + "&" # Adjust indent to match original, but ensure & is at column 6 if len(indent) > 5: cont_indent = indent[:5] + "&" + indent[5:] second_part = cont_indent + line[72:] return [first_part, second_part] return None def split_comma_list(line): """Split line at last comma before column 72.""" max_col = 72 # First check if any commas exist before max_col comma_positions = [i for i, c in enumerate(line[:max_col]) if c == ',' and not in_string_at_pos(line, i)] if not comma_positions: return None split_pos = comma_positions[-1] + 1 # Split after comma first_part = line[:split_pos].rstrip() + "&\n" indent = line[:line.find(line.lstrip())] cont_indent = " " *5 + "&" if len(indent) >5: cont_indent = indent[:5] + "&" + indent[5:] second_part = cont_indent + line[split_pos:] return [first_part, second_part] def in_string_at_pos(line, pos): """Check if position pos is inside a quoted string.""" in_string = False current_quote = None idx =0 while idx < pos: if line[idx] in "'\"": if not in_string: in_string = True current_quote = line[idx] idx = find_quote_end(line, idx) else: if line[idx] == current_quote: if idx +1 < len(line) and line[idx+1] == current_quote: idx +=2 continue in_string = False current_quote = None idx +=1 return in_string def split_arithmetic_expression(line): """Split line at last arithmetic operator before column 72.""" max_col =72 # Define operators in order of priority (longer operators first) operators = ['.GE.', '.LE.', '.GT.', '.LT.', '.EQ.', '.NE.', '.AND.', '.OR.', '.NOT.', '=', '+', '-', '*', '/'] split_pos = -1 # Check each operator from end of the max_col range backwards for op in operators: op_len = len(op) idx = max_col - op_len while idx >=0: if line[idx:idx+op_len] == op and not in_string_at_pos(line, idx): # Ensure operator is not part of a variable name left_ok = idx ==0 or not line[idx-1].isalnum() and line[idx-1] != '_' right_ok = (idx + op_len >= len(line)) or not line[idx+op_len].isalnum() and line[idx+op_len] != '_' if left_ok and right_ok: split_pos = idx + op_len break idx -=1 if split_pos != -1: break if split_pos == -1: return None first_part = line[:split_pos].rstrip() + "&\n" indent = line[:line.find(line.lstrip())] cont_indent = " " *5 + "&" if len(indent) >5: cont_indent = indent[:5] + "&" + indent[5:] second_part = cont_indent + line[split_pos:] return [first_part, second_part] def process_long_line(line): """Process a long line according to the splitting rules.""" if len(line) <=72: return [line] # Try string split first result = split_string_line(line) if result: return result # Try comma list split result = split_comma_list(line) if result: return result # Try arithmetic expression split result = split_arithmetic_expression(line) if result: return result # Fallback: split at column 72 if no other rules apply first_part = line[:72].rstrip() + "&\n" indent = line[:line.find(line.lstrip())] cont_indent = " " *5 + "&" if len(indent) >5: cont_indent = indent[:5] + "&" + indent[5:] second_part = cont_indent + line[72:] return [first_part, second_part] def process_file(filename): """Process the file, splitting long lines and writing output.""" try: with open(filename, "r") as infile, open(f"{filename}_fixed.f", "w") as outfile: for line_number, line in enumerate(infile, start=1): if is_comment_line(line): outfile.write(line) continue processed_lines = process_long_line(line) for pline in processed_lines: outfile.write(pline) print(f"Processing complete. Fixed file saved as {filename}_fixed.f") except FileNotFoundError: print("File not found. Please check the file path and try again.") except Exception as e: print(f"An error occurred: {e}") if __name__ == "__main__": if len(sys.argv) !=2: print("Usage: python script.py filename") else: filename = sys.argv[1] process_file(filename)
脚本说明
- 新增
is_comment_line函数判断是否为注释行,跳过处理 find_quote_end和in_string_at_pos函数处理字符串识别,解决引号嵌套和转义问题- 三个拆分函数分别对应三种场景,按优先级依次尝试
- 拆分后的子行自动处理缩进,确保
&位于第6列 - 处理后的文件保存为原文件名加
_fixed.f后缀
内容的提问来源于stack exchange,提问作者Foad S. Farimani
相关产品推荐
相关产品推荐

