使用Boost Spirit X3解析HTML遇C1202编译错误求助
Boost Spirit X3 HTML解析器递归依赖编译错误解决
问题详情
使用Boost Spirit X3编写HTML解析器时触发编译错误:
fatal error C1202: recursive type or function dependency context too complex
已知错误源于html_element_与tag_block_解析器的互相递归引用,但官方Qi/X3示例无法覆盖当前复杂场景,原代码如下:
#include <boost/spirit/home/x3.hpp> #include <boost/fusion/include/adapt_struct.hpp> #include <boost/spirit/home/x3/support/ast/position_tagged.hpp> #include <boost/spirit/home/x3/support/ast/variant.hpp> #include <iostream> using namespace boost::spirit::x3; struct tag_name{}; struct html_tag; struct html_comment; struct attribute_data : boost::spirit::x3::position_tagged { std::string name; boost::optional<std::string> value; }; struct tag_header : boost::spirit::x3::position_tagged { std::string name; std::vector<attribute_data> attributes; }; struct self_tag: boost::spirit::x3::position_tagged { tag_header header; }; struct html_element : boost::spirit::x3::position_tagged, boost::spirit::x3::variant< std::string, self_tag, boost::recursive_wrapper<html_tag>>{ using base_type::base_type; using base_type::operator=; }; struct html_tag: boost::spirit::x3::position_tagged { tag_header header; std::vector<html_element> children; }; BOOST_FUSION_ADAPT_STRUCT(attribute_data, name, value); BOOST_FUSION_ADAPT_STRUCT(tag_header, name, attributes); BOOST_FUSION_ADAPT_STRUCT(self_tag, header); BOOST_FUSION_ADAPT_STRUCT(html_tag,header,children); // These are the attributes parser, seems fine struct attribute_parser_id; auto attribute_identifier_= rule<attribute_parser_id, std::string>{"AttributeIdentifier"} = lexeme[+(char_ - char_(" /=>"))]; auto attribute_value_= rule<attribute_parser_id, std::string>{"AttributeValue"} = lexeme["\"" > +(char_ - char_("\"")) > "\""]|lexeme["'" > +(char_ - char_("'")) > "'"]| lexeme[+(char_ - char_(" />"))]; auto single_attribute_ = rule<attribute_parser_id, attribute_data>{"SingleAttribute"} = attribute_identifier_ > -("=>" attribute_value_); auto attributes_ = rule<attribute_parser_id, std::vector<attribute_data>>{"Attributes"} = (*single_attribute_); struct tag_parser_id; auto tag_name_begin_func = [](auto &ctx){ get<tag_name>(ctx) = _attr(ctx).name; //_val(ctx).header.name = _attr(ctx); std::cout << typeid(_val(ctx)).name() << std::endl; }; auto tag_name_end_func = [](auto &ctx){ _pass(ctx) = get<tag_name>(ctx) == _attr(ctx); }; auto self_tag_name_action = [](auto &ctx){ _val(ctx).header.name = _attr(ctx); }; auto self_tag_attribute_action = [](auto &ctx){ _val(ctx).header.attributes = _attr(ctx); }; auto inner_text = lexeme[+(char_-'<')]; auto tag_name_ = rule<tag_parser_id, std::string>{"HtmlTagName"} = lexeme[*(char_ - char_(" />"))]; auto self_tag_ = rule<tag_parser_id, self_tag>{"HtmlSelfTag"} = '<' > tag_name_[self_tag_name_action] > attributes_[self_tag_attribute_action] > "/>"; auto tag_header_ = rule<tag_parser_id, tag_header>{"HtmlTagBlockHeader"} = '<' > tag_name_ > attributes_ > '>'; rule<tag_parser_id, html_tag> tag_block_; rule<tag_parser_id, html_element> html_element_ = "HtmlElement"; auto tag_block__def = with<tag_name>(std::string())[tag_header_[tag_name_begin_func] > (*html_element_) > "</" > omit[tag_name_[tag_name_end_func]] > '>']; auto html_element__def = inner_text | self_tag_ | tag_block_ ; BOOST_SPIRIT_DEFINE(tag_block_, html_element_); int main() { std::string source = "<div data-src=\"https://www.google.com\" id='hello world'></div>"; html_element result; auto const parser = html_element_; auto parse_result = phrase_parse(source.begin(), source.end(), parser, ascii::space, result); }
解决方法
问题核心在于递归解析器规则的声明与初始化方式错误,X3处理递归规则时需确保规则仅先声明、后定义,避免编译器推导类型时陷入无限递归。具体修改点:
- 调整递归规则声明:将
html_element_的初始化改为默认构造,避免提前绑定字符串干扰类型推导:rule<tag_parser_id, html_element> html_element_; // 仅声明,不初始化 - 优化规则定义顺序:确保所有依赖子规则声明完成后,再通过
BOOST_SPIRIT_DEFINE绑定递归规则的定义。 - 移除调试输出:暂时删除
tag_name_begin_func中的std::cout,减少编译上下文复杂度。
修改后的完整代码
#include <boost/spirit/home/x3.hpp> #include <boost/fusion/include/adapt_struct.hpp> #include <boost/spirit/home/x3/support/ast/position_tagged.hpp> #include <boost/spirit/home/x3/support/ast/variant.hpp> #include <iostream> using namespace boost::spirit::x3; struct tag_name{}; struct html_tag; struct html_comment; struct attribute_data : boost::spirit::x3::position_tagged { std::string name; boost::optional<std::string> value; }; struct tag_header : boost::spirit::x3::position_tagged { std::string name; std::vector<attribute_data> attributes; }; struct self_tag : boost::spirit::x3::position_tagged { tag_header header; }; struct html_element : boost::spirit::x3::position_tagged, boost::spirit::x3::variant<std::string, self_tag, boost::recursive_wrapper<html_tag>> { using base_type::base_type; using base_type::operator=; }; struct html_tag : boost::spirit::x3::position_tagged { tag_header header; std::vector<html_element> children; }; BOOST_FUSION_ADAPT_STRUCT(attribute_data, name, value); BOOST_FUSION_ADAPT_STRUCT(tag_header, name, attributes); BOOST_FUSION_ADAPT_STRUCT(self_tag, header); BOOST_FUSION_ADAPT_STRUCT(html_tag, header, children); // 属性解析器部分 struct attribute_parser_id; auto attribute_identifier_ = rule<attribute_parser_id, std::string>{"AttributeIdentifier"} = lexeme[+(char_ - char_(" /=>"))]; auto attribute_value_ = rule<attribute_parser_id, std::string>{"AttributeValue"} = lexeme["\"" > +(char_ - char_("\"")) > "\""] | lexeme["'" > +(char_ - char_("'")) > "'"] | lexeme[+(char_ - char_(" />"))]; auto single_attribute_ = rule<attribute_parser_id, attribute_data>{"SingleAttribute"} = attribute_identifier_ > -('=' > attribute_value_); // 修正HTML属性赋值符为'='而非"=>" auto attributes_ = rule<attribute_parser_id, std::vector<attribute_data>>{"Attributes"} = *single_attribute_; // 标签解析器部分 struct tag_parser_id; auto tag_name_begin_func = [](auto &ctx) { get<tag_name>(ctx) = _attr(ctx).name; }; auto tag_name_end_func = [](auto &ctx) { _pass(ctx) = get<tag_name>(ctx) == _attr(ctx); }; auto self_tag_name_action = [](auto &ctx) { _val(ctx).header.name = _attr(ctx); }; auto self_tag_attribute_action = [](auto &ctx) { _val(ctx).header.attributes = _attr(ctx); }; auto inner_text = lexeme[+(char_ - '<')]; auto tag_name_ = rule<tag_parser_id, std::string>{"HtmlTagName"} = lexeme[+(char_ - char_(" />"))]; // 修正标签名解析规则,禁止空标签名 auto self_tag_ = rule<tag_parser_id, self_tag>{"HtmlSelfTag"} = '<' > tag_name_[self_tag_name_action] > attributes_[self_tag_attribute_action] > "/>"; auto tag_header_ = rule<tag_parser_id, tag_header>{"HtmlTagBlockHeader"} = '<' > tag_name_ > attributes_ > '>'; // 递归规则仅声明 rule<tag_parser_id, html_tag> tag_block_; rule<tag_parser_id, html_element> html_element_; // 定义递归规则 auto tag_block__def = with<tag_name>(std::string())[tag_header_[tag_name_begin_func] > (*html_element_) > "</" > omit[tag_name_[tag_name_end_func]] > '>']; auto html_element__def = inner_text | self_tag_ | tag_block_; BOOST_SPIRIT_DEFINE(tag_block_, html_element_) int main() { std::string source = "<div data-src=\"https://www.google.com\" id='hello world'></div>"; html_element result; auto const parser = html_element_; auto parse_result = phrase_parse(source.begin(), source.end(), parser, ascii::space, result); std::cout << (parse_result ? "解析成功" : "解析失败") << std::endl; return 0; }
额外修复说明
除解决编译问题外,还修正了两处逻辑错误:
- 将属性赋值运算符从
"=>"改为HTML标准的'=' - 标签名解析规则从
*(char_ ...)改为+(char_ ...),禁止空标签名
内容的提问来源于stack exchange,提问作者Hackman Lo
相关产品推荐
相关产品推荐

