You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Elasticsearch如何将terms聚合的Key传入子聚合的Script中?

问题描述

我将每场游戏的结果存储为一个文档,玩家及其得分分别存储在users和scores数组中,示例数据如下:

[
  {
    "gameId": "game01",
    "users": [
      "user01",
      "user02"
    ],
    "@timestamp": "2022-08-11T17:00:00.000Z",
    "scores": [
      4,
      1
    ]
  },
  {
    "gameId": "game02",
    "users": [
      "user01",
      "user02"
    ],
    "@timestamp": "2022-08-12T17:00:00.000Z",
    "scores": [
      3,
      1
    ]
  },
  {
    "gameId": "game02",
    "users": [
      "user02",
      "user03"
    ],
    "@timestamp": "2022-08-12T18:00:00.000Z",
    "scores": [
      2,
      4
    ]
  }
]

我需要按日期→游戏→用户的层级聚合,计算每个用户的每日总得分,预期结果如下:

{
  "aggregations": {
    "aggByDate": {
      "buckets": [
        {
          "key_as_string": "2022-08-11T00:00:00.000+08:00",
          "doc_count": 1,
          "aggByGame": {
            "buckets": [
              {
                "key": "game01",
                "doc_count": 1,
                "aggByUser": {
                  "buckets": [
                    {
                      "key": "user01",
                      "doc_count": 1,
                      "totalScore": {
                        "value": 4
                      }
                    },
                    {
                      "key": "user02",
                      "doc_count": 1,
                      "totalScore": {
                        "value": 1
                      }
                    }
                  ]
                }
              }
            ]
          }
        },
        {
          "key_as_string": "2022-08-12T00:00:00.000+08:00",
          "doc_count": 2,
          "aggByGame": {
            "buckets": [
              {
                "key": "game02",
                "doc_count": 1,
                "aggByUser": {
                  "buckets": [
                    {
                      "key": "user01",
                      "doc_count": 1,
                      "totalScore": {
                        "value": 3
                      }
                    },
                    {
                      "key": "user02",
                      "doc_count": 2,
                      "totalScore": {
                        "value": 3
                      }
                    },
                    {
                      "key": "user03",
                      "doc_count": 1,
                      "totalScore": {
                        "value": 4
                      }
                    }
                  ]
                }
              }
            ]
          }
        }
      ]
    }
  }
}

但尝试的查询中,无法在子聚合的脚本里获取当前用户桶的key(即目标用户ID),导致无法匹配对应的得分值。使用的是Elasticsearch 7.10版本。

解决方法

方法一:调整数据结构(推荐,性能更优)

将users和scores两个平行数组改成嵌套对象数组,直接利用嵌套聚合关联用户与得分,无需脚本匹配索引。

1. 修改索引映射

更新索引的映射,添加players嵌套字段:

PUT /games/_mapping
{
  "properties": {
    "players": {
      "type": "nested",
      "properties": {
        "user": {"type": "keyword"},
        "score": {"type": "integer"}
      }
    },
    "gameId": {"type": "keyword"},
    "@timestamp": {"type": "date"}
  }
}

2. 重新导入数据

将原数据转换为嵌套结构,示例:

[
  {
    "gameId": "game01",
    "@timestamp": "2022-08-11T17:00:00.000Z",
    "players": [
      {"user": "user01", "score": 4},
      {"user": "user02", "score": 1}
    ]
  },
  {
    "gameId": "game02",
    "@timestamp": "2022-08-12T17:00:00.000Z",
    "players": [
      {"user": "user01", "score": 3},
      {"user": "user02", "score": 1}
    ]
  },
  {
    "gameId": "game02",
    "@timestamp": "2022-08-12T18:00:00.000Z",
    "players": [
      {"user": "user02", "score": 2},
      {"user": "user03", "score": 4}
    ]
  }
]

3. 执行聚合查询

使用嵌套聚合实现需求:

{
  "size": 0,
  "aggs": {
    "aggByDate": {
      "date_histogram": {
        "field": "@timestamp",
        "interval": "1d",
        "time_zone": "+8",
        "min_doc_count": 1
      },
      "aggs": {
        "aggByGame": {
          "terms": {
            "field": "gameId"
          },
          "aggs": {
            "nested_players": {
              "nested": {
                "path": "players"
              },
              "aggs": {
                "aggByUser": {
                  "terms": {
                    "field": "players.user"
                  },
                  "aggs": {
                    "totalScore": {
                      "sum": {
                        "field": "players.score"
                      }
                    }
                  }
                }
              }
            }
          }
        }
      }
    }
  }
}

该查询会直接返回符合预期的结果,且性能远优于脚本方式。

方法二:在脚本中获取桶key(无需修改数据结构)

在Elasticsearch 7.10中,子聚合的脚本可以通过params._bucket_key获取当前terms桶的key值,修改原查询的脚本部分即可:

{
  "size": 0,
  "aggs": {
    "aggByDate": {
      "date_histogram": {
        "field": "@timestamp",
        "interval": "1d",
        "time_zone": "+8",
        "min_doc_count": 1
      },
      "aggs": {
        "aggByGame": {
          "terms": {
            "field": "gameId"
          },
          "aggs": {
            "aggByUser": {
              "terms": {
                "field": "users"
              },
              "aggs": {
                "totalScore": {
                  "sum": {
                    "script": {
                      "source": """
                        String targetUser = params._bucket_key;
                        int idx = doc['users'].values.indexOf(targetUser);
                        return idx != -1 ? doc['scores'].values[idx] : 0;
                      """
                    }
                  }
                }
              }
            }
          }
        }
      }
    }
  }
}

注意:这种方式依赖数组索引的严格对应,且脚本执行会带来一定性能开销,数据量较大时不推荐。


内容的提问来源于stack exchange,提问作者WeiJun

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.18 21:35:23