Delphi抓取HTML出现标签换行拆分,如何格式化得到完整标签?
解决方法
实现需求分为两个核心步骤:先修复被拆分的破损标签,再对完整HTML做规范格式化。
步骤1:修复拆分标签
所有包含在< >符号之间的换行符、回车符均为非法拆分,直接清除即可解决标签被拆成两行的问题。
步骤2:格式化HTML
在每个开闭标签的前后插入对应换行,再按节点层级加缩进,即可得到按节点换行的规范效果。
完整代码实现
工具函数
// 修复被拆分的破损标签 function FixBrokenTags(const RawHTML: string): string; var I: Integer; InTag: Boolean; begin InTag := False; Result := ''; for I := 1 to Length(RawHTML) do begin if RawHTML[I] = '<' then InTag := True; if RawHTML[I] = '>' then InTag := False; // 跳过标签内部的换行、回车字符 if not (InTag and (RawHTML[I] in [#10, #13])) then Result := Result + RawHTML[I]; end; end; // 按节点规范格式化HTML function FormatHTML(const RawHTML: string): string; const INDENT_SIZE = 2; // 缩进空格数可自行调整 var I: Integer; InTag, InText, InComment: Boolean; IndentLevel: Integer; CurrentChar: Char; Indent: string; begin Result := ''; InTag := False; InText := False; InComment := False; IndentLevel := 0; for I := 1 to Length(RawHTML) do begin CurrentChar := RawHTML[I]; // 过滤注释内容不做格式化处理 if (I <= Length(RawHTML) - 3) and (Copy(RawHTML, I, 4) = '<!--') then InComment := True; if (I >= 4) and (Copy(RawHTML, I-3, 4) = '-->') then InComment := False; if InComment then begin Result := Result + CurrentChar; Continue; end; if CurrentChar = '<' then begin InTag := True; if InText then begin Result := Result + #13#10; InText := False; end; // 闭标签先减缩进 if (I < Length(RawHTML)) and (RawHTML[I+1] = '/') then begin Dec(IndentLevel); Indent := StringOfChar(' ', IndentLevel * INDENT_SIZE); Result := Result + Indent; end else begin // 开标签加缩进 Indent := StringOfChar(' ', IndentLevel * INDENT_SIZE); Result := Result + Indent; end; Result := Result + CurrentChar; end else if CurrentChar = '>' then begin InTag := False; Result := Result + CurrentChar + #13#10; // 自闭合标签、声明标签不改变缩进层级 if not ((I > 1) and (RawHTML[I-1] = '/')) and not ((I > 1) and (RawHTML[I-1] = '?')) then begin if (I < Length(RawHTML)) and (RawHTML[I+1] <> '<') then InText := True else if not ((I <= Length(RawHTML)) and (RawHTML[I+1] = '/')) then Inc(IndentLevel); end; end else begin if not InTag then InText := True; Result := Result + CurrentChar; end; end; // 清除多余空行 Result := StringReplace(Result, #13#10#13#10, #13#10, [rfReplaceAll]); end;
调用示例
var Cli: THTTPClient; RawHTML, FixedHTML, FormattedHTML: string; begin Cli := THTTPClient.Create; try RawHTML := Cli.Get(URL).ContentAsString; FixedHTML := FixBrokenTags(RawHTML); FormattedHTML := FormatHTML(FixedHTML); Memo1.Text := FormattedHTML; finally Cli.Free; end; end;
注意事项
如果需要保留script、style标签内的原有格式,可在格式化逻辑中增加对应标签的判断,跳过标签内部内容的格式化处理即可。
内容的提问来源于stack exchange,提问作者Bald
相关产品推荐
相关产品推荐

