You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Spring Boot 3中Elasticsearch自定义xyz_analyzer在dev环境找不到

Spring Boot 3 + Elasticsearch 自定义分词器xyz_analyzerDev环境无法识别问题

问题现象

  • 本地环境:自定义分词器xyz_analyzer可正常识别,测试请求返回正确分词结果
  • Dev环境:转发ES Pod后执行相同请求,提示failed to find analyzer [xyz_analyzer],但abc_analyzer在所有环境均能正常识别

本地测试成功示例

请求:

POST http://localhost:9200/efg.efgdata-000001/_analyze

请求体:

{
    "analyzer": "xyz_analyzer",
    "text": "GmbH %26"
}

响应:

{
    "tokens": [
        {
            "token": "gmbh",
            "start_offset": 0,
            "end_offset": 4,
            "type": "word",
            "position": 0
        },
        {
            "token": "%26",
            "start_offset": 5,
            "end_offset": 8,
            "type": "word",
            "position": 1
        }
    ]
}

Dev环境失败示例

响应:

{
    "error": {
        "root_cause": [
            {
                "type": "illegal_argument_exception",
                "reason": "failed to find analyzer [xyz_analyzer]"
            }
        ],
        "type": "illegal_argument_exception",
        "reason": "failed to find analyzer [xyz_analyzer]"
    },
    "status": 400
}

相关配置

自定义分词器配置类

import org.apache.lucene.analysis.core.LowerCaseFilterFactory;
import org.apache.lucene.analysis.standard.StandardTokenizerFactory;
import org.hibernate.search.backend.elasticsearch.analysis.ElasticsearchAnalysisConfigurationContext;
import org.hibernate.search.backend.elasticsearch.analysis.ElasticsearchAnalysisConfigurer;
import org.hibernate.search.backend.lucene.analysis.LuceneAnalysisConfigurationContext;
import org.hibernate.search.backend.lucene.analysis.LuceneAnalysisConfigurer;
import org.springframework.context.annotation.Configuration;

@Configuration
public class CustomAnalysisConfigurer implements ElasticsearchAnalysisConfigurer, LuceneAnalysisConfigurer {

    public static final String ABC_ANALYZER = "abc_analyzer";

    public static final String XYZ_ANALYZER = "xyz_analyzer";

    @Override
    public void configure( ElasticsearchAnalysisConfigurationContext context ) {
        // filter section
        context.tokenFilter( "remove_whitespace" )
                .type( "pattern_replace" )
                .param( "pattern", "\\s+" )
                .param( "replacement", "" );

        context.tokenFilter( "remove_dash" )
                .type( "pattern_replace" )
                .param( "pattern", "-" )
                .param( "replacement", "" );

        // analyzer section
        context.analyzer( ABC_ANALYZER )
                .custom()
                .tokenizer( "keyword" )
                .tokenFilters( "lowercase", "remove_whitespace", "remove_dash" );

        context.tokenizer( "special_char_tokenizer" )
                .type( "whitespace" )
                .param( "token_chars", "letter", "digit", "symbol", "punctuation" );

        context.analyzer( XYZ_ANALYZER )
                .custom()
                .tokenizer( "special_char_tokenizer" )
                .tokenFilters( "lowercase" );

    }

    @Override
    public void configure( LuceneAnalysisConfigurationContext context ) {
        context.analyzer( ABC_ANALYZER )
                .custom()
                .tokenizer( StandardTokenizerFactory.class )
                .tokenFilter( LowerCaseFilterFactory.class );

        context.analyzer( XYZ_ANALYZER )
                .custom()
                .tokenizer( "special_char_tokenizer" )
                .tokenFilter( LowerCaseFilterFactory.class );
    }
}

Elasticsearch分析器JSON配置

{
  "filter": {
    "remove_dash": {
      "pattern": "-",
      "type": "pattern_replace",
      "replacement": ""
    },
    "remove_whitespace": {
      "pattern": "\\s+",
      "type": "pattern_replace",
      "replacement": ""
    }
  },
  "tokenizer": {
    "special_char_tokenizer": {
      "type": "keyword",
      "filter": [
        "token_chars", "letter", "digit", "symbol", "punctuation"
      ]
    }
  },
  "analyzer": {
    "abc_analyzer": {
      "filter": [
        "lowercase",
        "remove_whitespace",
        "remove_dash"
      ],
      "type": "custom",
      "tokenizer": "keyword"
    },
    "xyz_analyzer": {
      "filter": [
        "lowercase"
      ],
      "type": "custom",
      "tokenizer": "special_char_tokenizer"
    }
  }
}

索引映射配置

{
  "dynamic": "strict",
  "properties": {
    "_entity_type": {
      "type": "keyword",
      "index": false
    },
    "beginDate": {
      "type": "date",
      "doc_values": false,
      "format": "uuuu-MM-dd'T'HH:mm:ss.SSSSSSSSSZZZZZ"
    },
    "A": {
      "type": "text"
    },
    "B": {
      "type": "text"
    },
    "C": {
      "type": "text"
    },
    "D": {
      "type": "keyword",
      "doc_values": false
    },
    "E": {
      "dynamic": "strict",
      "properties": {
        "city": {
          "type": "text"
        },
        "F": {
          "type": "long",
          "doc_values": false
        },
        "name": {
          "type": "text",
          "analyzer": "xyz_analyzer"
        },
        "G": {
          "type": "text"
        },
        "H": {
          "type": "text"
        },
        "I": {
          "type": "text"
        },
        "J": {
          "type": "text"
        }
      }
    },
    "K": {
      "type": "long",
      "doc_values": false
    },
    "L": {
      "type": "date",
      "doc_values": false,
      "format": "uuuu-MM-dd'T'HH:mm:ss.SSSSSSSSSZZZZZ"
    },
    "M": {
      "type": "long",
      "doc_values": false
    },
    "ABCNumber": {
      "type": "text",
      "analyzer": "abc_analyzer"
    },
    "N": {
      "type": "text"
    },
    "O": {
      "type": "date",
      "doc_values": false,
      "format": "uuuu-MM-dd'T'HH:mm:ss.SSSSSSSSSZZZZZ"
    },
    "P": {
      "type": "long",
      "doc_values": false
    },
    "Q": {
      "type": "long",
      "doc_values": false
    },
    "R": {
      "type": "text"
    },
    "S": {
      "type": "text"
    },
    "T": {
      "type": "text"
    },
    "U": {
      "type": "text"
    }
  }
}

实体类字段配置

@FullTextField( searchable = Searchable.YES, projectable = Projectable.YES, analyzer = ABC_ANALYZER )
@Size( max = 20 )
@Column( name = "ABC_NUMBER" )
private String ABCNumber;

@FullTextField( searchable = Searchable.YES, projectable = Projectable.YES, analyzer = XYZ_ANALYZER )
@Column( name = "NAME" )
private String name;

环境索引设置对比

本地环境索引设置

执行请求:GET http://localhost:9200/efg.efgdata-000001/_settings
响应:

{
    "efg.efgdata-000001": {
        "settings": {
            "index": {
                "routing": {
                    "allocation": {
                        "include": {
                            "_tier_preference": "data_content"
                        }
                    }
                },
                "number_of_shards": "1",
                "provided_name": "efg.efgdata-000001",
                "creation_date": "1696403353164",
                "analysis": {
                    "filter": {
                        "remove_dash": {
                            "pattern": "-",
                            "type": "pattern_replace",
                            "replacement": ""
                        },
                        "remove_whitespace": {
                            "pattern": "\\s+",
                            "type": "pattern_replace",
                            "replacement": ""
                        }
                    },
                    "analyzer": {
                        "abc_analyzer": {
                            "filter": [
                                "lowercase",
                                "remove_whitespace",
                                "remove_dash"
                            ],
                            "type": "custom",
                            "tokenizer": "keyword"
                        },
                        "xyz_analyzer": {
                            "filter": [
                                "lowercase"
                            ],
                            "type": "custom",
                            "tokenizer": "special_char_tokenizer"
                        }
                    },
                    "tokenizer": {
                        "special_char_tokenizer": {
                            "token_chars": [
                                "letter",
                                "digit",
                                "symbol",
                                "punctuation"
                            ],
                            "type": "whitespace"
                        }
                    }
                },
                "number_of_replicas": "1",
                "uuid": "G1XVBph-TeGd5RpfETKsFg",
                "version": {
                    "created": "7120099"
                }
            }
        }
    }
}

Dev环境索引设置

执行请求:GET http://localhost:9200/efg.efgdata-000001/_settings
响应:

{
    "efg.efgdata-000001": {
        "settings": {
            "index": {
                "routing": {
                    "allocation": {
                        "include": {
                            "_tier_preference": "data_content"
                        }
                    }
                },
                "number_of_shards": "1",
                "provided_name": "efg.efgdata-000001",
                "creation_date": "1689166583198",
                "analysis": {
                    "filter": {
                        "remove_dash": {
                            "pattern": "-",
                            "type": "pattern_replace",
                            "replacement": ""
                        },
                        "remove_whitespace": {
                            "pattern": "\\s+",
                            "type": "pattern_replace",
                            "replacement": ""
                        }
                    },
                    "analyzer": {
                        "abc_analyzer": {
                            "filter": [
                                "lowercase",
                                "remove_whitespace",
                                "remove_dash"
                            ],
                            "type": "custom",
                            "tokenizer": "keyword"
                        }
                    }
                },
                "number_of_replicas": "1",
                "uuid": "9gIpm08iRba32EDhgaaskA",
                "version": {
                    "created": "7171199"
                }
            }
        }
    }
}

问题分析与解决方案

问题根因

从索引设置对比可明确:Dev环境的索引配置缺失xyz_analyzer和special_char_tokenizer的定义,仅保留了abc_analyzer和相关过滤器。同时提供的JSON分析器配置中,special_char_tokenizer类型错误设置为keyword,与代码配置的whitespace不一致,可能导致同步失败。

解决方案

  1. 重建Dev环境索引:Elasticsearch无法在索引创建后修改分析器配置,需删除现有索引后重新创建,确保Hibernate Search将完整分析器配置同步到ES。
  2. 统一配置类型:将JSON配置中的special_char_tokenizer类型修正为whitespace,与代码配置保持一致,避免配置冲突。
  3. 检查环境配置:确认Dev环境的Hibernate Search开启了自动创建索引功能,无阻止新分析器同步的配置项。

内容的提问来源于stack exchange,提问作者tdog

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.09 09:52:02