You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

OpenSearch嵌套聚合:如何获取指定嵌套属性的分桶

版本

OpenSearch 2.13

问题描述

我正在尝试使用OpenSearch的搜索API和分桶聚合为索引生成面(Facet),需求是为满足特定条件的嵌套字段生成分桶聚合。

我的索引包含一个nested类型的properties字段(每个文档可包含多个独立查询的属性),还有type、createdBy等object类型字段(每个文档各有一个),映射结构如下:

"idx_container_view_list": {
    "mappings": {
        "properties": {
            "ancestorPath": {
                "type": "keyword"
            },
            "color": {
                "type": "keyword"
            },
            "coverFile": {
                "type": "keyword"
            },
            "createdAt": {
                "type": "date"
            },
            "createdBy": {
                "properties": {
                    "email": {
                        "type": "keyword"
                    },
                    "firstname": {
                        "type": "keyword"
                    },
                    "lastname": {
                        "type": "keyword"
                    },
                    "username": {
                        "type": "keyword"
                    },
                    "uuid": {
                        "type": "keyword"
                    }
                }
            },
            "deletedAt": {
                "type": "date"
            },
            "deletedBy": ...,
            "label": {
                "type": "keyword"
            },
            "parent": {
                "type": "keyword"
            },
            "properties": {
                "type": "nested",
                "properties": {
                    "propertyName": {
                        "type": "keyword"
                    },
                    "propertyUuid": {
                        "type": "keyword"
                    },
                    "propertyValue": {
                        "type": "keyword"
                    },
                    "schemaPropertyUuid": {
                        "type": "keyword"
                    },
                    "uuid": {
                        "type": "keyword"
                    },
                    "valueAsLocation": {
                        "properties": {
                            "latitude": {
                                "type": "long"
                            },
                            "longitude": {
                                "type": "long"
                            },
                            "placeName": {
                                "type": "text",
                                "fields": {
                                    "keyword": {
                                        "type": "keyword",
                                        "ignore_above": 256
                                    }
                                }
                            }
                        }
                    },
                    "valueAsNumber": {
                        "type": "float"
                    }
                }
            },
            "publishedAt": {
                "type": "date"
            },
            "publishedBy": ...
            },
            "shared": {
                "type": "boolean"
            },
            "type": {
                "properties": {
                    "name": {
                        "type": "keyword"
                    },
                    "slug": {
                        "type": "keyword"
                    },
                    "uuid": {
                        "type": "keyword"
                    }
                }
            },
            "updatedAt": {
                "type": "date"
            },
            "updatedBy": ...,
            "uuid": {
                "type": "keyword"
            }
        }
    }
}

我已实现所有属性的通用聚合查询(获取前20个属性的前5个值的计数):

"size" : 0,
"aggs": {
    "different_container_property_types": {
        "nested": {
            "path":"properties"
        },
        "aggs": {
            "property_uuid_bucket": {
                "terms": {
                    "field": "properties.propertyUuid",
                    "size": 20
                },
                "aggs": {
                    "property_value_bucket": {
                        "terms": {
                            "field": "properties.propertyValue",
                            "size": 5
                        }
                    }
                }
            }
        }
    }
}

但我需要仅针对特定propertyUuid生成分桶聚合。尝试添加嵌套查询过滤文档后,返回的是符合条件文档中所有属性的分桶,而非目标属性的单独分桶:

"size" : 0,
"query" : {
    "nested": {
        "path": "properties",
        "query":{
            "term": {
                "properties.propertyUuid": "62329d5b-dbc9-4022-93e7-4690c9069a7e"
            }
        }
    }
},
"aggs": {
    "different_container_property_types": {
        "nested": {
            "path":"properties"
        },
        "aggs": {
            "property_uuid_bucket": {
                "terms": {
                    "field": "properties.propertyUuid",
                    "size": 20
                },
                "aggs": {
                    "property_value_bucket": {
                        "terms": {
                            "field": "properties.propertyValue",
                            "size": 5
                        }
                    }
                }
            }
        }
    }
}

返回结果片段:

"took": 46,
"timed_out": false,
"_shards": {
    "total": 1,
    "successful": 1,
    "skipped": 0,
    "failed": 0
},
"hits": {
    "total": {
        "value": 352,
        "relation": "eq"
    },
    "max_score": null,
    "hits": []
},
"aggregations": {
    "different_container_property_types": {
        "doc_count": 7624,
        "property_uuid_bucket": {
            "doc_count_error_upper_bound": 0,
            "sum_other_doc_count": 583,
            "buckets": [
                {
                    "key": "02d39964-e894-4d8e-8fdd-32bf2a9c6eab",
                    "doc_count": 353,
                    "property_value_bucket": {
                        "doc_count_error_upper_bound": 0,
                        "sum_other_doc_count": 84,
                        "buckets": [...]
                    }
                },
                {
                    "key": "03df8c5e-61c1-4485-8978-dd2bb96c6790",
                    ...
                },
                {
                    "key": "62329d5b-dbc9-4022-93e7-4690c9069a7e",
                    "doc_count": 352,
                    "property_value_bucket": {
                        "doc_count_error_upper_bound": 0,
                        "sum_other_doc_count": 77,
                        "buckets": [
                            {
                                "key": "b0cf8292-6005-4095-946d-51fc908243b8",
                                "doc_count": 138
                            },
                            {
                                "key": "3f5230ab-8614-436b-abc3-d5cad7f4b5a7",
                                "doc_count": 75
                            },
                            {
                                "key": "25135d79-a4c1-44a9-beed-b926b94a40fb",
                                "doc_count": 34
                            },
                            {
                                "key": "ab2807d8-1946-40e4-a156-cf3195438697",
                                "doc_count": 31
                            },
                            {
                                "key": "d03982e8-872b-4f02-a6b8-fa461e16389e",
                                "doc_count": 24
                            }
                        ]
                    }
                },
               ...

我理解这是因为OpenSearch为每个嵌套属性生成隐藏文档,过滤出352个包含目标propertyUuid的文档后,聚合包含这些文档中的全部7624个属性分桶。现询问是否有方法仅获取指定propertyUuid的分桶聚合,以避免不必要的集群负载。

解决方案

要仅针对指定propertyUuid生成分桶聚合,需要在嵌套聚合内部添加过滤条件,直接限制聚合只处理目标嵌套属性,避免遍历无关的嵌套文档,降低集群负载。以下是两种可行实现方式:

方式一:嵌套聚合内使用filter聚合

在嵌套聚合中先过滤出目标propertyUuid,再对其值进行分桶,同时顶层查询保留对含目标属性文档的过滤,进一步缩小处理范围:

"size": 0,
"query": {
    "nested": {
        "path": "properties",
        "query": {
            "term": {
                "properties.propertyUuid": "62329d5b-dbc9-4022-93e7-4690c9069a7e"
            }
        }
    }
},
"aggs": {
    "target_property": {
        "nested": {
            "path": "properties"
        },
        "aggs": {
            "filter_target_uuid": {
                "filter": {
                    "term": {
                        "properties.propertyUuid": "62329d5b-dbc9-4022-93e7-4690c9069a7e"
                    }
                },
                "aggs": {
                    "property_value_bucket": {
                        "terms": {
                            "field": "properties.propertyValue",
                            "size": 5
                        }
                    }
                }
            }
        }
    }
}

方式二:terms聚合使用include参数

如果不需要过滤文档,仅需统计目标propertyUuid的属性值,可直接在terms聚合中用include参数精确匹配目标UUID,省去顶层查询步骤:

"size": 0,
"aggs": {
    "target_property": {
        "nested": {
            "path": "properties"
        },
        "aggs": {
            "property_uuid_bucket": {
                "terms": {
                    "field": "properties.propertyUuid",
                    "include": "62329d5b-dbc9-4022-93e7-4690c9069a7e",
                    "size": 1
                },
                "aggs": {
                    "property_value_bucket": {
                        "terms": {
                            "field": "properties.propertyValue",
                            "size": 5
                        }
                    }
                }
            }
        }
    }
}

方案说明

  • 方式一适合需先筛选出含目标属性的文档,再聚合该属性值的场景,双重过滤能最大程度减少集群处理的数据量。
  • 方式二更轻量化,无需顶层查询,直接锁定目标UUID进行聚合,性能更优,适合仅需目标属性值统计的场景。

内容的提问来源于stack exchange,提问作者fyts

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.25 00:44:55