Spring Boot 3中Elasticsearch自定义xyz_analyzer在dev环境找不到
Spring Boot 3 + Elasticsearch 自定义分词器
xyz_analyzerDev环境无法识别问题 问题现象
- 本地环境:自定义分词器
xyz_analyzer可正常识别,测试请求返回正确分词结果 - Dev环境:转发ES Pod后执行相同请求,提示
failed to find analyzer [xyz_analyzer],但abc_analyzer在所有环境均能正常识别
本地测试成功示例
请求:
POST http://localhost:9200/efg.efgdata-000001/_analyze
请求体:
{ "analyzer": "xyz_analyzer", "text": "GmbH %26" }
响应:
{ "tokens": [ { "token": "gmbh", "start_offset": 0, "end_offset": 4, "type": "word", "position": 0 }, { "token": "%26", "start_offset": 5, "end_offset": 8, "type": "word", "position": 1 } ] }
Dev环境失败示例
响应:
{ "error": { "root_cause": [ { "type": "illegal_argument_exception", "reason": "failed to find analyzer [xyz_analyzer]" } ], "type": "illegal_argument_exception", "reason": "failed to find analyzer [xyz_analyzer]" }, "status": 400 }
相关配置
自定义分词器配置类
import org.apache.lucene.analysis.core.LowerCaseFilterFactory; import org.apache.lucene.analysis.standard.StandardTokenizerFactory; import org.hibernate.search.backend.elasticsearch.analysis.ElasticsearchAnalysisConfigurationContext; import org.hibernate.search.backend.elasticsearch.analysis.ElasticsearchAnalysisConfigurer; import org.hibernate.search.backend.lucene.analysis.LuceneAnalysisConfigurationContext; import org.hibernate.search.backend.lucene.analysis.LuceneAnalysisConfigurer; import org.springframework.context.annotation.Configuration; @Configuration public class CustomAnalysisConfigurer implements ElasticsearchAnalysisConfigurer, LuceneAnalysisConfigurer { public static final String ABC_ANALYZER = "abc_analyzer"; public static final String XYZ_ANALYZER = "xyz_analyzer"; @Override public void configure( ElasticsearchAnalysisConfigurationContext context ) { // filter section context.tokenFilter( "remove_whitespace" ) .type( "pattern_replace" ) .param( "pattern", "\\s+" ) .param( "replacement", "" ); context.tokenFilter( "remove_dash" ) .type( "pattern_replace" ) .param( "pattern", "-" ) .param( "replacement", "" ); // analyzer section context.analyzer( ABC_ANALYZER ) .custom() .tokenizer( "keyword" ) .tokenFilters( "lowercase", "remove_whitespace", "remove_dash" ); context.tokenizer( "special_char_tokenizer" ) .type( "whitespace" ) .param( "token_chars", "letter", "digit", "symbol", "punctuation" ); context.analyzer( XYZ_ANALYZER ) .custom() .tokenizer( "special_char_tokenizer" ) .tokenFilters( "lowercase" ); } @Override public void configure( LuceneAnalysisConfigurationContext context ) { context.analyzer( ABC_ANALYZER ) .custom() .tokenizer( StandardTokenizerFactory.class ) .tokenFilter( LowerCaseFilterFactory.class ); context.analyzer( XYZ_ANALYZER ) .custom() .tokenizer( "special_char_tokenizer" ) .tokenFilter( LowerCaseFilterFactory.class ); } }
Elasticsearch分析器JSON配置
{ "filter": { "remove_dash": { "pattern": "-", "type": "pattern_replace", "replacement": "" }, "remove_whitespace": { "pattern": "\\s+", "type": "pattern_replace", "replacement": "" } }, "tokenizer": { "special_char_tokenizer": { "type": "keyword", "filter": [ "token_chars", "letter", "digit", "symbol", "punctuation" ] } }, "analyzer": { "abc_analyzer": { "filter": [ "lowercase", "remove_whitespace", "remove_dash" ], "type": "custom", "tokenizer": "keyword" }, "xyz_analyzer": { "filter": [ "lowercase" ], "type": "custom", "tokenizer": "special_char_tokenizer" } } }
索引映射配置
{ "dynamic": "strict", "properties": { "_entity_type": { "type": "keyword", "index": false }, "beginDate": { "type": "date", "doc_values": false, "format": "uuuu-MM-dd'T'HH:mm:ss.SSSSSSSSSZZZZZ" }, "A": { "type": "text" }, "B": { "type": "text" }, "C": { "type": "text" }, "D": { "type": "keyword", "doc_values": false }, "E": { "dynamic": "strict", "properties": { "city": { "type": "text" }, "F": { "type": "long", "doc_values": false }, "name": { "type": "text", "analyzer": "xyz_analyzer" }, "G": { "type": "text" }, "H": { "type": "text" }, "I": { "type": "text" }, "J": { "type": "text" } } }, "K": { "type": "long", "doc_values": false }, "L": { "type": "date", "doc_values": false, "format": "uuuu-MM-dd'T'HH:mm:ss.SSSSSSSSSZZZZZ" }, "M": { "type": "long", "doc_values": false }, "ABCNumber": { "type": "text", "analyzer": "abc_analyzer" }, "N": { "type": "text" }, "O": { "type": "date", "doc_values": false, "format": "uuuu-MM-dd'T'HH:mm:ss.SSSSSSSSSZZZZZ" }, "P": { "type": "long", "doc_values": false }, "Q": { "type": "long", "doc_values": false }, "R": { "type": "text" }, "S": { "type": "text" }, "T": { "type": "text" }, "U": { "type": "text" } } }
实体类字段配置
@FullTextField( searchable = Searchable.YES, projectable = Projectable.YES, analyzer = ABC_ANALYZER ) @Size( max = 20 ) @Column( name = "ABC_NUMBER" ) private String ABCNumber; @FullTextField( searchable = Searchable.YES, projectable = Projectable.YES, analyzer = XYZ_ANALYZER ) @Column( name = "NAME" ) private String name;
环境索引设置对比
本地环境索引设置
执行请求:GET http://localhost:9200/efg.efgdata-000001/_settings
响应:
{ "efg.efgdata-000001": { "settings": { "index": { "routing": { "allocation": { "include": { "_tier_preference": "data_content" } } }, "number_of_shards": "1", "provided_name": "efg.efgdata-000001", "creation_date": "1696403353164", "analysis": { "filter": { "remove_dash": { "pattern": "-", "type": "pattern_replace", "replacement": "" }, "remove_whitespace": { "pattern": "\\s+", "type": "pattern_replace", "replacement": "" } }, "analyzer": { "abc_analyzer": { "filter": [ "lowercase", "remove_whitespace", "remove_dash" ], "type": "custom", "tokenizer": "keyword" }, "xyz_analyzer": { "filter": [ "lowercase" ], "type": "custom", "tokenizer": "special_char_tokenizer" } }, "tokenizer": { "special_char_tokenizer": { "token_chars": [ "letter", "digit", "symbol", "punctuation" ], "type": "whitespace" } } }, "number_of_replicas": "1", "uuid": "G1XVBph-TeGd5RpfETKsFg", "version": { "created": "7120099" } } } } }
Dev环境索引设置
执行请求:GET http://localhost:9200/efg.efgdata-000001/_settings
响应:
{ "efg.efgdata-000001": { "settings": { "index": { "routing": { "allocation": { "include": { "_tier_preference": "data_content" } } }, "number_of_shards": "1", "provided_name": "efg.efgdata-000001", "creation_date": "1689166583198", "analysis": { "filter": { "remove_dash": { "pattern": "-", "type": "pattern_replace", "replacement": "" }, "remove_whitespace": { "pattern": "\\s+", "type": "pattern_replace", "replacement": "" } }, "analyzer": { "abc_analyzer": { "filter": [ "lowercase", "remove_whitespace", "remove_dash" ], "type": "custom", "tokenizer": "keyword" } } }, "number_of_replicas": "1", "uuid": "9gIpm08iRba32EDhgaaskA", "version": { "created": "7171199" } } } } }
问题分析与解决方案
问题根因
从索引设置对比可明确:Dev环境的索引配置缺失xyz_analyzer和special_char_tokenizer的定义,仅保留了abc_analyzer和相关过滤器。同时提供的JSON分析器配置中,special_char_tokenizer类型错误设置为keyword,与代码配置的whitespace不一致,可能导致同步失败。
解决方案
- 重建Dev环境索引:Elasticsearch无法在索引创建后修改分析器配置,需删除现有索引后重新创建,确保Hibernate Search将完整分析器配置同步到ES。
- 统一配置类型:将JSON配置中的
special_char_tokenizer类型修正为whitespace,与代码配置保持一致,避免配置冲突。 - 检查环境配置:确认Dev环境的Hibernate Search开启了自动创建索引功能,无阻止新分析器同步的配置项。
内容的提问来源于stack exchange,提问作者tdog
相关产品推荐
相关产品推荐

